@inproceedings{amaral-cejas-etal-2026-large,
title = "Large-Scale Multilingual {SMS} Fraud Detection For Telecom Networks",
author = "Amaral Cejas, Orlando and
Guo, Yuejun and
Tang, Qiang",
editor = "Mitkov, Ruslan and
Mu{\~n}oz, Rafael and
Lloret, Elena and
Ranasinghe, Tharindu and
Estevanell-Valladares, Ernesto L. and
Lamsiyah, Salima and
Montoyo, Andr{\'e}s and
Ezzini, Saad",
booktitle = "Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security",
month = jun,
year = "2026",
address = "Alicante, Spain",
publisher = "Department of Languages and Information Systems, University of Alicante",
url = "https://aclanthology.org/2026.nlpaics-1.7/",
pages = "64--73",
abstract = "Short Message Service is a fundamental communication channel in modern telecom networks, yet its ubiquity continues to be exploited for large-scale attacks. Recent advances in multilingual embeddings and large language models have shown strong performance on general text classification tasks. However, their effectiveness and efficiency for multilingual SMS fraud detection under constraints of real telecom environments, remain underexplored. In particular, existing studies largely focus on monolingual datasets, cloud-based inference, or overlook the constraints imposed by real telecom environments. In this work, we investigate large-scale multilingual SMS classification for HAM, SPAM, and SMISHING detection using embedding models. We construct and curate a proprietary multilingual SMS dataset and conduct a systematic evaluation of four different multilingual embedding models. Using the obtained dataset, we fine-tune the models, demonstrating that domain-adapted embeddings significantly improve SMS classification across several languages. Overall, this study addresses the gap between embedding-centric NLP research and real-world telecom requirements, providing empirical and practical insights for effective and efficient deployment of multilingual SMS fraud detection systems. We open source all non-proprietary material at: https://figshare.com/s/1df3ca8d08a4eb4d2712."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="amaral-cejas-etal-2026-large">
<titleInfo>
<title>Large-Scale Multilingual SMS Fraud Detection For Telecom Networks</title>
</titleInfo>
<name type="personal">
<namePart type="given">Orlando</namePart>
<namePart type="family">Amaral Cejas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yuejun</namePart>
<namePart type="family">Guo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Qiang</namePart>
<namePart type="family">Tang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ruslan</namePart>
<namePart type="family">Mitkov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rafael</namePart>
<namePart type="family">Muñoz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Lloret</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tharindu</namePart>
<namePart type="family">Ranasinghe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ernesto</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Estevanell-Valladares</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Salima</namePart>
<namePart type="family">Lamsiyah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andrés</namePart>
<namePart type="family">Montoyo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Department of Languages and Information Systems, University of Alicante</publisher>
<place>
<placeTerm type="text">Alicante, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Short Message Service is a fundamental communication channel in modern telecom networks, yet its ubiquity continues to be exploited for large-scale attacks. Recent advances in multilingual embeddings and large language models have shown strong performance on general text classification tasks. However, their effectiveness and efficiency for multilingual SMS fraud detection under constraints of real telecom environments, remain underexplored. In particular, existing studies largely focus on monolingual datasets, cloud-based inference, or overlook the constraints imposed by real telecom environments. In this work, we investigate large-scale multilingual SMS classification for HAM, SPAM, and SMISHING detection using embedding models. We construct and curate a proprietary multilingual SMS dataset and conduct a systematic evaluation of four different multilingual embedding models. Using the obtained dataset, we fine-tune the models, demonstrating that domain-adapted embeddings significantly improve SMS classification across several languages. Overall, this study addresses the gap between embedding-centric NLP research and real-world telecom requirements, providing empirical and practical insights for effective and efficient deployment of multilingual SMS fraud detection systems. We open source all non-proprietary material at: https://figshare.com/s/1df3ca8d08a4eb4d2712.</abstract>
<identifier type="citekey">amaral-cejas-etal-2026-large</identifier>
<location>
<url>https://aclanthology.org/2026.nlpaics-1.7/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>64</start>
<end>73</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Large-Scale Multilingual SMS Fraud Detection For Telecom Networks
%A Amaral Cejas, Orlando
%A Guo, Yuejun
%A Tang, Qiang
%Y Mitkov, Ruslan
%Y Muñoz, Rafael
%Y Lloret, Elena
%Y Ranasinghe, Tharindu
%Y Estevanell-Valladares, Ernesto L.
%Y Lamsiyah, Salima
%Y Montoyo, Andrés
%Y Ezzini, Saad
%S Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security
%D 2026
%8 June
%I Department of Languages and Information Systems, University of Alicante
%C Alicante, Spain
%F amaral-cejas-etal-2026-large
%X Short Message Service is a fundamental communication channel in modern telecom networks, yet its ubiquity continues to be exploited for large-scale attacks. Recent advances in multilingual embeddings and large language models have shown strong performance on general text classification tasks. However, their effectiveness and efficiency for multilingual SMS fraud detection under constraints of real telecom environments, remain underexplored. In particular, existing studies largely focus on monolingual datasets, cloud-based inference, or overlook the constraints imposed by real telecom environments. In this work, we investigate large-scale multilingual SMS classification for HAM, SPAM, and SMISHING detection using embedding models. We construct and curate a proprietary multilingual SMS dataset and conduct a systematic evaluation of four different multilingual embedding models. Using the obtained dataset, we fine-tune the models, demonstrating that domain-adapted embeddings significantly improve SMS classification across several languages. Overall, this study addresses the gap between embedding-centric NLP research and real-world telecom requirements, providing empirical and practical insights for effective and efficient deployment of multilingual SMS fraud detection systems. We open source all non-proprietary material at: https://figshare.com/s/1df3ca8d08a4eb4d2712.
%U https://aclanthology.org/2026.nlpaics-1.7/
%P 64-73
Markdown (Informal)
[Large-Scale Multilingual SMS Fraud Detection For Telecom Networks](https://aclanthology.org/2026.nlpaics-1.7/) (Amaral Cejas et al., NLPAICS 2026)
ACL
- Orlando Amaral Cejas, Yuejun Guo, and Qiang Tang. 2026. Large-Scale Multilingual SMS Fraud Detection For Telecom Networks. In Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security, pages 64–73, Alicante, Spain. Department of Languages and Information Systems, University of Alicante.