@inproceedings{kany-etal-2026-automatic,
title = "Automatic Prediction of Child Speech Fluency with Game-Based Data from {G}erman Preschoolers",
author = {Kany, Valentin and
M{\"o}bius, Bernd and
Trouvain, J{\"u}rgen},
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.439/",
doi = "10.63317/48vj8xeqn5ok",
pages = "5607--5616",
abstract = "This paper introduces an approach to automatically predict the speech fluency of preschool children as part of Language Proficiency Assessments. We use spontaneous speech data from children with German as native and second language aged 4{--}6 years, collected via a game{--}based elicitation method. The recordings were mainly annotated manually on various fluency-related phenomena. The resulting feature values were compared to human fluency ratings of the same data. The human ratings and the fluency-related acoustic features were used to build Cumulative Link Mixed Models (CLMMs) with and without splines to test their ability to predict the human ratings with multiple metrics (Spearman{'}s {\ensuremath{\rho}}, MAE, quadratic weighted {\ensuremath{\kappa}}). Results show that a parsimonious linear model already reaches near-human agreement (quadratic weighted kappa {\ensuremath{\kappa}} = 0.65) and that incorporating non-linear spline effects does not improve predictive accuracy. These findings suggest that relatively simple CLMMs can substitute additional human raters in fine-grained fluency assessment of preschool children, which is a task that is already challenging for trained listeners."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kany-etal-2026-automatic">
<titleInfo>
<title>Automatic Prediction of Child Speech Fluency with Game-Based Data from German Preschoolers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Valentin</namePart>
<namePart type="family">Kany</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Bernd</namePart>
<namePart type="family">Möbius</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jürgen</namePart>
<namePart type="family">Trouvain</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper introduces an approach to automatically predict the speech fluency of preschool children as part of Language Proficiency Assessments. We use spontaneous speech data from children with German as native and second language aged 4–6 years, collected via a game–based elicitation method. The recordings were mainly annotated manually on various fluency-related phenomena. The resulting feature values were compared to human fluency ratings of the same data. The human ratings and the fluency-related acoustic features were used to build Cumulative Link Mixed Models (CLMMs) with and without splines to test their ability to predict the human ratings with multiple metrics (Spearman’s \ensuremathρ, MAE, quadratic weighted \ensuremathąppa). Results show that a parsimonious linear model already reaches near-human agreement (quadratic weighted kappa \ensuremathąppa = 0.65) and that incorporating non-linear spline effects does not improve predictive accuracy. These findings suggest that relatively simple CLMMs can substitute additional human raters in fine-grained fluency assessment of preschool children, which is a task that is already challenging for trained listeners.</abstract>
<identifier type="citekey">kany-etal-2026-automatic</identifier>
<identifier type="doi">10.63317/48vj8xeqn5ok</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.439/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>5607</start>
<end>5616</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Automatic Prediction of Child Speech Fluency with Game-Based Data from German Preschoolers
%A Kany, Valentin
%A Möbius, Bernd
%A Trouvain, Jürgen
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F kany-etal-2026-automatic
%X This paper introduces an approach to automatically predict the speech fluency of preschool children as part of Language Proficiency Assessments. We use spontaneous speech data from children with German as native and second language aged 4–6 years, collected via a game–based elicitation method. The recordings were mainly annotated manually on various fluency-related phenomena. The resulting feature values were compared to human fluency ratings of the same data. The human ratings and the fluency-related acoustic features were used to build Cumulative Link Mixed Models (CLMMs) with and without splines to test their ability to predict the human ratings with multiple metrics (Spearman’s \ensuremathρ, MAE, quadratic weighted \ensuremathąppa). Results show that a parsimonious linear model already reaches near-human agreement (quadratic weighted kappa \ensuremathąppa = 0.65) and that incorporating non-linear spline effects does not improve predictive accuracy. These findings suggest that relatively simple CLMMs can substitute additional human raters in fine-grained fluency assessment of preschool children, which is a task that is already challenging for trained listeners.
%R 10.63317/48vj8xeqn5ok
%U https://aclanthology.org/2026.lrec-1.439/
%U https://doi.org/10.63317/48vj8xeqn5ok
%P 5607-5616
Markdown (Informal)
[Automatic Prediction of Child Speech Fluency with Game-Based Data from German Preschoolers](https://aclanthology.org/2026.lrec-1.439/) (Kany et al., LREC 2026)
ACL