@inproceedings{beccaria-etal-2026-automatic,
title = "Automatic Detection of Direct and Self-Repetitions in Naturalistic Speech Recordings of {F}rench- and {D}utch-Speaking Autistic Children",
author = "Beccaria, Federica and
Kolenberg, Marie and
Labendzki, Pierre and
Zink, Inge and
Kissine, Mikhail",
editor = {Kokkinakis, Dimitrios and
Themistocleous, Charalambos and
Dias, Ga{\"e}l and
Fraser, Kathleen C. and
{\"O}hman, Fredrik and
Pais, Sebasti{\~a}o},
booktitle = "Proceedings of the Sixth Resources and {P}rocess{I}ng of linguistic, para-linguistic and extra-linguistic Data from people with various forms of cognitive/psychiatric/developmental impairments in cooperation with the {MENTAL}.ai consortium",
month = may,
year = "2026",
address = "Palma, Mallorca, Spain",
publisher = "European Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.rapid-1.12/",
doi = "10.63317/4eo4uey3z8kj",
pages = "146--156",
abstract = "This study investigates the use of cosine similarity measures across syntactic, lexical, and semantic vector repre- sentations to detect repetitions in the spontaneous speech of autistic children. It focuses on direct repetitions (i.e., immediate verbatim repetitions of linguistic output produced by another individual) and self-repetitions (i.e., within-speaker recurrence). The performance of similarity-based methods is then compared with state-of-the-art black-box classification models based on BERT, trained on the same data. Using spontaneous speech data from French- and Dutch- speaking autistic children, the results show that lexical and semantic similarity provide reliable cues for identifying self-repetitions, achieving high precision and recall, with F1-scores exceeding 83{\%}, comparable to those obtained by BERT-based models. In contrast, direct repetitions are more difficult to detect using similarity-based approaches, with BERT models clearly outperforming them and reaching F1-scores above 73{\%}. Across all conditions, syntactic similarity consistently underperforms relative to lexical and semantic measures. These findings highlight the strengths and limitations of similarity-based approaches and suggest directions for future research, particularly in improving the detection of direct repetitions and assessing the cross-linguistic generalizability of these methods."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="beccaria-etal-2026-automatic">
<titleInfo>
<title>Automatic Detection of Direct and Self-Repetitions in Naturalistic Speech Recordings of French- and Dutch-Speaking Autistic Children</title>
</titleInfo>
<name type="personal">
<namePart type="given">Federica</namePart>
<namePart type="family">Beccaria</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marie</namePart>
<namePart type="family">Kolenberg</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pierre</namePart>
<namePart type="family">Labendzki</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Inge</namePart>
<namePart type="family">Zink</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mikhail</namePart>
<namePart type="family">Kissine</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Sixth Resources and ProcessIng of linguistic, para-linguistic and extra-linguistic Data from people with various forms of cognitive/psychiatric/developmental impairments in cooperation with the MENTAL.ai consortium</title>
</titleInfo>
<name type="personal">
<namePart type="given">Dimitrios</namePart>
<namePart type="family">Kokkinakis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Charalambos</namePart>
<namePart type="family">Themistocleous</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gaël</namePart>
<namePart type="family">Dias</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kathleen</namePart>
<namePart type="given">C</namePart>
<namePart type="family">Fraser</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fredrik</namePart>
<namePart type="family">Öhman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sebastião</namePart>
<namePart type="family">Pais</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This study investigates the use of cosine similarity measures across syntactic, lexical, and semantic vector repre- sentations to detect repetitions in the spontaneous speech of autistic children. It focuses on direct repetitions (i.e., immediate verbatim repetitions of linguistic output produced by another individual) and self-repetitions (i.e., within-speaker recurrence). The performance of similarity-based methods is then compared with state-of-the-art black-box classification models based on BERT, trained on the same data. Using spontaneous speech data from French- and Dutch- speaking autistic children, the results show that lexical and semantic similarity provide reliable cues for identifying self-repetitions, achieving high precision and recall, with F1-scores exceeding 83%, comparable to those obtained by BERT-based models. In contrast, direct repetitions are more difficult to detect using similarity-based approaches, with BERT models clearly outperforming them and reaching F1-scores above 73%. Across all conditions, syntactic similarity consistently underperforms relative to lexical and semantic measures. These findings highlight the strengths and limitations of similarity-based approaches and suggest directions for future research, particularly in improving the detection of direct repetitions and assessing the cross-linguistic generalizability of these methods.</abstract>
<identifier type="citekey">beccaria-etal-2026-automatic</identifier>
<identifier type="doi">10.63317/4eo4uey3z8kj</identifier>
<location>
<url>https://aclanthology.org/2026.rapid-1.12/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>146</start>
<end>156</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Automatic Detection of Direct and Self-Repetitions in Naturalistic Speech Recordings of French- and Dutch-Speaking Autistic Children
%A Beccaria, Federica
%A Kolenberg, Marie
%A Labendzki, Pierre
%A Zink, Inge
%A Kissine, Mikhail
%Y Kokkinakis, Dimitrios
%Y Themistocleous, Charalambos
%Y Dias, Gaël
%Y Fraser, Kathleen C.
%Y Öhman, Fredrik
%Y Pais, Sebastião
%S Proceedings of the Sixth Resources and ProcessIng of linguistic, para-linguistic and extra-linguistic Data from people with various forms of cognitive/psychiatric/developmental impairments in cooperation with the MENTAL.ai consortium
%D 2026
%8 May
%I European Language Resources Association (ELRA)
%C Palma, Mallorca, Spain
%F beccaria-etal-2026-automatic
%X This study investigates the use of cosine similarity measures across syntactic, lexical, and semantic vector repre- sentations to detect repetitions in the spontaneous speech of autistic children. It focuses on direct repetitions (i.e., immediate verbatim repetitions of linguistic output produced by another individual) and self-repetitions (i.e., within-speaker recurrence). The performance of similarity-based methods is then compared with state-of-the-art black-box classification models based on BERT, trained on the same data. Using spontaneous speech data from French- and Dutch- speaking autistic children, the results show that lexical and semantic similarity provide reliable cues for identifying self-repetitions, achieving high precision and recall, with F1-scores exceeding 83%, comparable to those obtained by BERT-based models. In contrast, direct repetitions are more difficult to detect using similarity-based approaches, with BERT models clearly outperforming them and reaching F1-scores above 73%. Across all conditions, syntactic similarity consistently underperforms relative to lexical and semantic measures. These findings highlight the strengths and limitations of similarity-based approaches and suggest directions for future research, particularly in improving the detection of direct repetitions and assessing the cross-linguistic generalizability of these methods.
%R 10.63317/4eo4uey3z8kj
%U https://aclanthology.org/2026.rapid-1.12/
%U https://doi.org/10.63317/4eo4uey3z8kj
%P 146-156
Markdown (Informal)
[Automatic Detection of Direct and Self-Repetitions in Naturalistic Speech Recordings of French- and Dutch-Speaking Autistic Children](https://aclanthology.org/2026.rapid-1.12/) (Beccaria et al., RaPID 2026)
ACL