@inproceedings{stein-2026-chat,
title = "From {CHAT} to Coded {C}o{NLL}-{U}: A Reproducible Pipeline for the Syntactic Annotation and Querying of Child Language Data",
author = "Stein, Achim",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.901/",
doi = "10.63317/498zo5heasd5",
pages = "11516--11523",
abstract = "The CHILDES database is a core resource for language acquisition research, yet its CHAT format poses significant challenges for modern computational analysis. To address this, we present a reproducible, open-source pipeline that transforms CHAT transcripts into annotated tabular (CSV) and CoNLL-U formats. Its core script, childes.py, automates the conversion and integrates part-of-speech tagging and dependency parsing. A key innovation is dql.py, a tool that uses a Grew dependency query language to systematically add user-defined linguistic codings to the parsed data. While the script is parametrised for various languages, the pipeline{'}s utility is demonstrated by applying it to the French CHILDES corpus to conduct a large-scale analysis of object clitic production. The resulting structured data reveals clear developmental trajectories, such as the gradual convergence of children{'}s dative clitic usage towards the adult input. The workflow and the resources it generates facilitate reproducible, data-driven research in language acquisition."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="stein-2026-chat">
<titleInfo>
<title>From CHAT to Coded CoNLL-U: A Reproducible Pipeline for the Syntactic Annotation and Querying of Child Language Data</title>
</titleInfo>
<name type="personal">
<namePart type="given">Achim</namePart>
<namePart type="family">Stein</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The CHILDES database is a core resource for language acquisition research, yet its CHAT format poses significant challenges for modern computational analysis. To address this, we present a reproducible, open-source pipeline that transforms CHAT transcripts into annotated tabular (CSV) and CoNLL-U formats. Its core script, childes.py, automates the conversion and integrates part-of-speech tagging and dependency parsing. A key innovation is dql.py, a tool that uses a Grew dependency query language to systematically add user-defined linguistic codings to the parsed data. While the script is parametrised for various languages, the pipeline’s utility is demonstrated by applying it to the French CHILDES corpus to conduct a large-scale analysis of object clitic production. The resulting structured data reveals clear developmental trajectories, such as the gradual convergence of children’s dative clitic usage towards the adult input. The workflow and the resources it generates facilitate reproducible, data-driven research in language acquisition.</abstract>
<identifier type="citekey">stein-2026-chat</identifier>
<identifier type="doi">10.63317/498zo5heasd5</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.901/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>11516</start>
<end>11523</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T From CHAT to Coded CoNLL-U: A Reproducible Pipeline for the Syntactic Annotation and Querying of Child Language Data
%A Stein, Achim
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F stein-2026-chat
%X The CHILDES database is a core resource for language acquisition research, yet its CHAT format poses significant challenges for modern computational analysis. To address this, we present a reproducible, open-source pipeline that transforms CHAT transcripts into annotated tabular (CSV) and CoNLL-U formats. Its core script, childes.py, automates the conversion and integrates part-of-speech tagging and dependency parsing. A key innovation is dql.py, a tool that uses a Grew dependency query language to systematically add user-defined linguistic codings to the parsed data. While the script is parametrised for various languages, the pipeline’s utility is demonstrated by applying it to the French CHILDES corpus to conduct a large-scale analysis of object clitic production. The resulting structured data reveals clear developmental trajectories, such as the gradual convergence of children’s dative clitic usage towards the adult input. The workflow and the resources it generates facilitate reproducible, data-driven research in language acquisition.
%R 10.63317/498zo5heasd5
%U https://aclanthology.org/2026.lrec-1.901/
%U https://doi.org/10.63317/498zo5heasd5
%P 11516-11523
Markdown (Informal)
[From CHAT to Coded CoNLL-U: A Reproducible Pipeline for the Syntactic Annotation and Querying of Child Language Data](https://aclanthology.org/2026.lrec-1.901/) (Stein, LREC 2026)
ACL