@inproceedings{sousa-silva-2026-detection,
title = "From Detection to Attribution: Forensic Linguistics and Adversarial Red Teaming as Complementary Responses to {LLM} Misuse",
author = "Sousa-Silva, Rui",
editor = "Mitkov, Ruslan and
Mu{\~n}oz, Rafael and
Lloret, Elena and
Ranasinghe, Tharindu and
Estevanell-Valladares, Ernesto L. and
Lamsiyah, Salima and
Montoyo, Andr{\'e}s and
Ezzini, Saad",
booktitle = "Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security",
month = jun,
year = "2026",
address = "Alicante, Spain",
publisher = "Department of Languages and Information Systems, University of Alicante",
url = "https://aclanthology.org/2026.nlpaics-1.13/",
pages = "122--133",
abstract = "The proliferation of Large Language Models (LLMs) has enabled the automation of cyber-attacks (including phishing, social engineering, and impersonation) at unprecedented scale, while existing safeguards remain routinely circumvented. Current detection approaches, predominantly based on stylometric and machine learning methods, face fundamental limitations against adaptive adversaries and struggle with the implicit, contextual, and pragmatic dimensions of language. This article proposes forensic linguistic analysis grounded in the theory of idiolect as a complementary approach to LLM-generated text detection and attribution. We adopt a red teaming methodology to generate synthetic toxic texts that bypass model guardrails to create controlled conditions for testing whether qualitative forensic analysis can succeed where quantitative approaches falter. The findings of our stylometric, character n-gram, and cluster analysis converge to provide moderate evidence that stylometric approaches succeed in discriminating authorship. However, they are not conclusive and hence fall short of current admissibility criteria across diverse jurisdictions. The article thus concludes that idiolect-based forensic analysis can distinguish genuine authorship from LLM-generated impersonation, even when surface features are manipulated. We discuss implications for legal and investigative contexts, where interpretable, theoretically-grounded expert analysis is required over black-box classifier outputs."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="sousa-silva-2026-detection">
<titleInfo>
<title>From Detection to Attribution: Forensic Linguistics and Adversarial Red Teaming as Complementary Responses to LLM Misuse</title>
</titleInfo>
<name type="personal">
<namePart type="given">Rui</namePart>
<namePart type="family">Sousa-Silva</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ruslan</namePart>
<namePart type="family">Mitkov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rafael</namePart>
<namePart type="family">Muñoz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Lloret</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tharindu</namePart>
<namePart type="family">Ranasinghe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ernesto</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Estevanell-Valladares</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Salima</namePart>
<namePart type="family">Lamsiyah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andrés</namePart>
<namePart type="family">Montoyo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Department of Languages and Information Systems, University of Alicante</publisher>
<place>
<placeTerm type="text">Alicante, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The proliferation of Large Language Models (LLMs) has enabled the automation of cyber-attacks (including phishing, social engineering, and impersonation) at unprecedented scale, while existing safeguards remain routinely circumvented. Current detection approaches, predominantly based on stylometric and machine learning methods, face fundamental limitations against adaptive adversaries and struggle with the implicit, contextual, and pragmatic dimensions of language. This article proposes forensic linguistic analysis grounded in the theory of idiolect as a complementary approach to LLM-generated text detection and attribution. We adopt a red teaming methodology to generate synthetic toxic texts that bypass model guardrails to create controlled conditions for testing whether qualitative forensic analysis can succeed where quantitative approaches falter. The findings of our stylometric, character n-gram, and cluster analysis converge to provide moderate evidence that stylometric approaches succeed in discriminating authorship. However, they are not conclusive and hence fall short of current admissibility criteria across diverse jurisdictions. The article thus concludes that idiolect-based forensic analysis can distinguish genuine authorship from LLM-generated impersonation, even when surface features are manipulated. We discuss implications for legal and investigative contexts, where interpretable, theoretically-grounded expert analysis is required over black-box classifier outputs.</abstract>
<identifier type="citekey">sousa-silva-2026-detection</identifier>
<location>
<url>https://aclanthology.org/2026.nlpaics-1.13/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>122</start>
<end>133</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T From Detection to Attribution: Forensic Linguistics and Adversarial Red Teaming as Complementary Responses to LLM Misuse
%A Sousa-Silva, Rui
%Y Mitkov, Ruslan
%Y Muñoz, Rafael
%Y Lloret, Elena
%Y Ranasinghe, Tharindu
%Y Estevanell-Valladares, Ernesto L.
%Y Lamsiyah, Salima
%Y Montoyo, Andrés
%Y Ezzini, Saad
%S Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security
%D 2026
%8 June
%I Department of Languages and Information Systems, University of Alicante
%C Alicante, Spain
%F sousa-silva-2026-detection
%X The proliferation of Large Language Models (LLMs) has enabled the automation of cyber-attacks (including phishing, social engineering, and impersonation) at unprecedented scale, while existing safeguards remain routinely circumvented. Current detection approaches, predominantly based on stylometric and machine learning methods, face fundamental limitations against adaptive adversaries and struggle with the implicit, contextual, and pragmatic dimensions of language. This article proposes forensic linguistic analysis grounded in the theory of idiolect as a complementary approach to LLM-generated text detection and attribution. We adopt a red teaming methodology to generate synthetic toxic texts that bypass model guardrails to create controlled conditions for testing whether qualitative forensic analysis can succeed where quantitative approaches falter. The findings of our stylometric, character n-gram, and cluster analysis converge to provide moderate evidence that stylometric approaches succeed in discriminating authorship. However, they are not conclusive and hence fall short of current admissibility criteria across diverse jurisdictions. The article thus concludes that idiolect-based forensic analysis can distinguish genuine authorship from LLM-generated impersonation, even when surface features are manipulated. We discuss implications for legal and investigative contexts, where interpretable, theoretically-grounded expert analysis is required over black-box classifier outputs.
%U https://aclanthology.org/2026.nlpaics-1.13/
%P 122-133
Markdown (Informal)
[From Detection to Attribution: Forensic Linguistics and Adversarial Red Teaming as Complementary Responses to LLM Misuse](https://aclanthology.org/2026.nlpaics-1.13/) (Sousa-Silva, NLPAICS 2026)
ACL