@inproceedings{marchenko-2026-well,
title = "How Well Do Commodity Text-to-Speech Systems Evade Acoustic Perturbation Detection? A Multi-Engine Evaluation Across 21 Languages",
author = "Marchenko, Anatoly",
editor = "Mitkov, Ruslan and
Mu{\~n}oz, Rafael and
Lloret, Elena and
Ranasinghe, Tharindu and
Estevanell-Valladares, Ernesto L. and
Lamsiyah, Salima and
Montoyo, Andr{\'e}s and
Ezzini, Saad",
booktitle = "Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security",
month = jun,
year = "2026",
address = "Alicante, Spain",
publisher = "Department of Languages and Information Systems, University of Alicante",
url = "https://aclanthology.org/2026.nlpaics-1.18/",
pages = "171--175",
abstract = "Jitter, shimmer, and harmonics-to-noise ratio (HNR) are often used to detect voice deepfakes, since these features capture biomechanical irregularities of vocal fold vibration that synthetic speech supposedly lacks. We test this assumption on three commodity TTS engines (Google TTS, Microsoft Edge TTS, macOS system voice) with 1,850 samples across 21 languages, measured against 29 emotion corpora in 24 languages (35,091 utterances). Three classifiers (logistic regression, SVM-RBF, Random Forest) all fail to reliably detect Edge TTS: the best result is F1 = 0.78. Effect sizes drop 2.1x-7.4x from Google TTS to Edge TTS. In ablation, no single feature exceeds F1 = 0.60 against Edge TTS. These three perturbation features, taken alone, can no longer separate commodity neural TTS from natural speech."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="marchenko-2026-well">
<titleInfo>
<title>How Well Do Commodity Text-to-Speech Systems Evade Acoustic Perturbation Detection? A Multi-Engine Evaluation Across 21 Languages</title>
</titleInfo>
<name type="personal">
<namePart type="given">Anatoly</namePart>
<namePart type="family">Marchenko</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ruslan</namePart>
<namePart type="family">Mitkov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rafael</namePart>
<namePart type="family">Muñoz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Lloret</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tharindu</namePart>
<namePart type="family">Ranasinghe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ernesto</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Estevanell-Valladares</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Salima</namePart>
<namePart type="family">Lamsiyah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andrés</namePart>
<namePart type="family">Montoyo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Department of Languages and Information Systems, University of Alicante</publisher>
<place>
<placeTerm type="text">Alicante, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Jitter, shimmer, and harmonics-to-noise ratio (HNR) are often used to detect voice deepfakes, since these features capture biomechanical irregularities of vocal fold vibration that synthetic speech supposedly lacks. We test this assumption on three commodity TTS engines (Google TTS, Microsoft Edge TTS, macOS system voice) with 1,850 samples across 21 languages, measured against 29 emotion corpora in 24 languages (35,091 utterances). Three classifiers (logistic regression, SVM-RBF, Random Forest) all fail to reliably detect Edge TTS: the best result is F1 = 0.78. Effect sizes drop 2.1x-7.4x from Google TTS to Edge TTS. In ablation, no single feature exceeds F1 = 0.60 against Edge TTS. These three perturbation features, taken alone, can no longer separate commodity neural TTS from natural speech.</abstract>
<identifier type="citekey">marchenko-2026-well</identifier>
<location>
<url>https://aclanthology.org/2026.nlpaics-1.18/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>171</start>
<end>175</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T How Well Do Commodity Text-to-Speech Systems Evade Acoustic Perturbation Detection? A Multi-Engine Evaluation Across 21 Languages
%A Marchenko, Anatoly
%Y Mitkov, Ruslan
%Y Muñoz, Rafael
%Y Lloret, Elena
%Y Ranasinghe, Tharindu
%Y Estevanell-Valladares, Ernesto L.
%Y Lamsiyah, Salima
%Y Montoyo, Andrés
%Y Ezzini, Saad
%S Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security
%D 2026
%8 June
%I Department of Languages and Information Systems, University of Alicante
%C Alicante, Spain
%F marchenko-2026-well
%X Jitter, shimmer, and harmonics-to-noise ratio (HNR) are often used to detect voice deepfakes, since these features capture biomechanical irregularities of vocal fold vibration that synthetic speech supposedly lacks. We test this assumption on three commodity TTS engines (Google TTS, Microsoft Edge TTS, macOS system voice) with 1,850 samples across 21 languages, measured against 29 emotion corpora in 24 languages (35,091 utterances). Three classifiers (logistic regression, SVM-RBF, Random Forest) all fail to reliably detect Edge TTS: the best result is F1 = 0.78. Effect sizes drop 2.1x-7.4x from Google TTS to Edge TTS. In ablation, no single feature exceeds F1 = 0.60 against Edge TTS. These three perturbation features, taken alone, can no longer separate commodity neural TTS from natural speech.
%U https://aclanthology.org/2026.nlpaics-1.18/
%P 171-175
Markdown (Informal)
[How Well Do Commodity Text-to-Speech Systems Evade Acoustic Perturbation Detection? A Multi-Engine Evaluation Across 21 Languages](https://aclanthology.org/2026.nlpaics-1.18/) (Marchenko, NLPAICS 2026)
ACL