@inproceedings{ponchard-serrano-2026-comparative,
title = "A Comparative Study of {P}arkinsonian Speech Corpora for Deep Learning-Based Detection of Dysarthria",
author = "Ponchard, Clara and
Serrano, Pierre",
editor = "Rapp, Reinhard and
Terryn, Ayla Rigouts and
Sharoff, Serge and
Zweigenbaum, Pierre",
booktitle = "Proceedings of the 19th Workshop on Building and Using Comparable Corpora ({BUCC})",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.bucc-1.2/",
doi = "10.63317/27zb48j5vv5f",
pages = "2--8",
abstract = "Idiopathic Parkinson{'}s disease is associated with motor speech impairments collectively referred to as hypokinetic dysarthria, which can appear at early disease stages and remain challenging to assess objectively in clinical practice. Most automatic assessment studies rely on individual speech corpora analyzed in isolation, leaving open questions regarding their comparability and their suitability for joint use within unified classification frameworks. This study explicitly investigates the cross-corpus comparability of existing Parkinsonian speech datasets designed for hypokinetic dysarthria assessment. Rather than assuming their compatibility, we evaluate it empirically through the generalization performance of classification systems trained on single or multiple corpora. We examine which datasets can be effectively combined and whether multi-corpus training improves robustness across heterogeneous recording conditions and speech tasks. Four corpora are evaluated under intra-corpus, cross-corpus, and out-of-domain settings. Results demonstrate that multi-corpus training enhances robustness and generalization performance, while also revealing substantial differences in cross-dataset compatibility. These findings provide a clearer understanding of the degree of comparability between existing resources and offer practical guidelines for the design of future corpora and more generalizable tools for the automatic clinical assessment of Parkinsonian speech."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="ponchard-serrano-2026-comparative">
<titleInfo>
<title>A Comparative Study of Parkinsonian Speech Corpora for Deep Learning-Based Detection of Dysarthria</title>
</titleInfo>
<name type="personal">
<namePart type="given">Clara</namePart>
<namePart type="family">Ponchard</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pierre</namePart>
<namePart type="family">Serrano</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 19th Workshop on Building and Using Comparable Corpora (BUCC)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Reinhard</namePart>
<namePart type="family">Rapp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ayla</namePart>
<namePart type="given">Rigouts</namePart>
<namePart type="family">Terryn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Serge</namePart>
<namePart type="family">Sharoff</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pierre</namePart>
<namePart type="family">Zweigenbaum</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Idiopathic Parkinson’s disease is associated with motor speech impairments collectively referred to as hypokinetic dysarthria, which can appear at early disease stages and remain challenging to assess objectively in clinical practice. Most automatic assessment studies rely on individual speech corpora analyzed in isolation, leaving open questions regarding their comparability and their suitability for joint use within unified classification frameworks. This study explicitly investigates the cross-corpus comparability of existing Parkinsonian speech datasets designed for hypokinetic dysarthria assessment. Rather than assuming their compatibility, we evaluate it empirically through the generalization performance of classification systems trained on single or multiple corpora. We examine which datasets can be effectively combined and whether multi-corpus training improves robustness across heterogeneous recording conditions and speech tasks. Four corpora are evaluated under intra-corpus, cross-corpus, and out-of-domain settings. Results demonstrate that multi-corpus training enhances robustness and generalization performance, while also revealing substantial differences in cross-dataset compatibility. These findings provide a clearer understanding of the degree of comparability between existing resources and offer practical guidelines for the design of future corpora and more generalizable tools for the automatic clinical assessment of Parkinsonian speech.</abstract>
<identifier type="citekey">ponchard-serrano-2026-comparative</identifier>
<identifier type="doi">10.63317/27zb48j5vv5f</identifier>
<location>
<url>https://aclanthology.org/2026.bucc-1.2/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>2</start>
<end>8</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T A Comparative Study of Parkinsonian Speech Corpora for Deep Learning-Based Detection of Dysarthria
%A Ponchard, Clara
%A Serrano, Pierre
%Y Rapp, Reinhard
%Y Terryn, Ayla Rigouts
%Y Sharoff, Serge
%Y Zweigenbaum, Pierre
%S Proceedings of the 19th Workshop on Building and Using Comparable Corpora (BUCC)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F ponchard-serrano-2026-comparative
%X Idiopathic Parkinson’s disease is associated with motor speech impairments collectively referred to as hypokinetic dysarthria, which can appear at early disease stages and remain challenging to assess objectively in clinical practice. Most automatic assessment studies rely on individual speech corpora analyzed in isolation, leaving open questions regarding their comparability and their suitability for joint use within unified classification frameworks. This study explicitly investigates the cross-corpus comparability of existing Parkinsonian speech datasets designed for hypokinetic dysarthria assessment. Rather than assuming their compatibility, we evaluate it empirically through the generalization performance of classification systems trained on single or multiple corpora. We examine which datasets can be effectively combined and whether multi-corpus training improves robustness across heterogeneous recording conditions and speech tasks. Four corpora are evaluated under intra-corpus, cross-corpus, and out-of-domain settings. Results demonstrate that multi-corpus training enhances robustness and generalization performance, while also revealing substantial differences in cross-dataset compatibility. These findings provide a clearer understanding of the degree of comparability between existing resources and offer practical guidelines for the design of future corpora and more generalizable tools for the automatic clinical assessment of Parkinsonian speech.
%R 10.63317/27zb48j5vv5f
%U https://aclanthology.org/2026.bucc-1.2/
%U https://doi.org/10.63317/27zb48j5vv5f
%P 2-8
Markdown (Informal)
[A Comparative Study of Parkinsonian Speech Corpora for Deep Learning-Based Detection of Dysarthria](https://aclanthology.org/2026.bucc-1.2/) (Ponchard & Serrano, BUCC 2026)
ACL