@inproceedings{morandini-2026-decision,
title = "From Decision Tree to Detection Pipeline: Formalizing van {D}ijk{'}s Socio-Cognitive Framework for Automated Anti-Language Identification in {RICO} Transcripts",
author = "Morandini, Elena",
editor = "Mitkov, Ruslan and
Mu{\~n}oz, Rafael and
Lloret, Elena and
Ranasinghe, Tharindu and
Estevanell-Valladares, Ernesto L. and
Lamsiyah, Salima and
Montoyo, Andr{\'e}s and
Ezzini, Saad",
booktitle = "Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security",
month = jun,
year = "2026",
address = "Alicante, Spain",
publisher = "Department of Languages and Information Systems, University of Alicante",
url = "https://aclanthology.org/2026.nlpaics-1.22/",
pages = "204--214",
abstract = "This paper proposes a context-first NLP detection pipeline for automated anti-language identification in RICO wiretap transcripts. Current threat-detection classifiers fail on organized crime discourse because criminal intent is encoded through implicature and relexicalization rather than explicit lexical markers. The pipeline addresses this architectural mismatch by formalizing van Dijk{'}s (2011) socio-cognitive CDA framework as a sequential seven-step decision tree mapped to concrete NLP subtasks: from speaker-role classification and genre detection to deontic feature extraction and ensemble scoring. A six-feature micro-level vector (F1{--}F6), validated against a 14,072-word corpus of authenticated Mafia communications, operationalizes the ideological square as a two-axis feature space that measures discursive distance between the ingroup and the outgroup. Preliminary evaluation confirms statistically significant patterns ({\ensuremath{\chi}}{\texttwosuperior} = 90.82, p {\ensuremath{<}} 0.001 for pragmatic divergence; 4:1 deontic saturation ratio) consistent with anti-language characteristics. The pipeline enables three LEA applications: automated flagging, context-sensitive decoding, and communication network analysis. Ethical considerations regarding false positives, privacy, and evidentiary standards are discussed."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="morandini-2026-decision">
<titleInfo>
<title>From Decision Tree to Detection Pipeline: Formalizing van Dijk’s Socio-Cognitive Framework for Automated Anti-Language Identification in RICO Transcripts</title>
</titleInfo>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Morandini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ruslan</namePart>
<namePart type="family">Mitkov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rafael</namePart>
<namePart type="family">Muñoz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Lloret</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tharindu</namePart>
<namePart type="family">Ranasinghe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ernesto</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Estevanell-Valladares</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Salima</namePart>
<namePart type="family">Lamsiyah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andrés</namePart>
<namePart type="family">Montoyo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Department of Languages and Information Systems, University of Alicante</publisher>
<place>
<placeTerm type="text">Alicante, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper proposes a context-first NLP detection pipeline for automated anti-language identification in RICO wiretap transcripts. Current threat-detection classifiers fail on organized crime discourse because criminal intent is encoded through implicature and relexicalization rather than explicit lexical markers. The pipeline addresses this architectural mismatch by formalizing van Dijk’s (2011) socio-cognitive CDA framework as a sequential seven-step decision tree mapped to concrete NLP subtasks: from speaker-role classification and genre detection to deontic feature extraction and ensemble scoring. A six-feature micro-level vector (F1–F6), validated against a 14,072-word corpus of authenticated Mafia communications, operationalizes the ideological square as a two-axis feature space that measures discursive distance between the ingroup and the outgroup. Preliminary evaluation confirms statistically significant patterns (\ensuremathχ² = 90.82, p \ensuremath< 0.001 for pragmatic divergence; 4:1 deontic saturation ratio) consistent with anti-language characteristics. The pipeline enables three LEA applications: automated flagging, context-sensitive decoding, and communication network analysis. Ethical considerations regarding false positives, privacy, and evidentiary standards are discussed.</abstract>
<identifier type="citekey">morandini-2026-decision</identifier>
<location>
<url>https://aclanthology.org/2026.nlpaics-1.22/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>204</start>
<end>214</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T From Decision Tree to Detection Pipeline: Formalizing van Dijk’s Socio-Cognitive Framework for Automated Anti-Language Identification in RICO Transcripts
%A Morandini, Elena
%Y Mitkov, Ruslan
%Y Muñoz, Rafael
%Y Lloret, Elena
%Y Ranasinghe, Tharindu
%Y Estevanell-Valladares, Ernesto L.
%Y Lamsiyah, Salima
%Y Montoyo, Andrés
%Y Ezzini, Saad
%S Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security
%D 2026
%8 June
%I Department of Languages and Information Systems, University of Alicante
%C Alicante, Spain
%F morandini-2026-decision
%X This paper proposes a context-first NLP detection pipeline for automated anti-language identification in RICO wiretap transcripts. Current threat-detection classifiers fail on organized crime discourse because criminal intent is encoded through implicature and relexicalization rather than explicit lexical markers. The pipeline addresses this architectural mismatch by formalizing van Dijk’s (2011) socio-cognitive CDA framework as a sequential seven-step decision tree mapped to concrete NLP subtasks: from speaker-role classification and genre detection to deontic feature extraction and ensemble scoring. A six-feature micro-level vector (F1–F6), validated against a 14,072-word corpus of authenticated Mafia communications, operationalizes the ideological square as a two-axis feature space that measures discursive distance between the ingroup and the outgroup. Preliminary evaluation confirms statistically significant patterns (\ensuremathχ² = 90.82, p \ensuremath< 0.001 for pragmatic divergence; 4:1 deontic saturation ratio) consistent with anti-language characteristics. The pipeline enables three LEA applications: automated flagging, context-sensitive decoding, and communication network analysis. Ethical considerations regarding false positives, privacy, and evidentiary standards are discussed.
%U https://aclanthology.org/2026.nlpaics-1.22/
%P 204-214
Markdown (Informal)
[From Decision Tree to Detection Pipeline: Formalizing van Dijk’s Socio-Cognitive Framework for Automated Anti-Language Identification in RICO Transcripts](https://aclanthology.org/2026.nlpaics-1.22/) (Morandini, NLPAICS 2026)
ACL