@inproceedings{vladu-etal-2026-building,
title = "Building Collaborative Speech Corpora for Low-Resource Languages: The {G}alician Dataset in Mozilla Common Voice",
author = "Vladu, Adina Ioana and
Fern{\'a}ndez Rei, Elisa and
P{\'e}rez Lago, Mar{\'i}a",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.135/",
doi = "10.63317/4gd79cq3cump",
pages = "1710--1720",
abstract = "This paper presents the methodology and outcomes of building collaborative speech corpora in Mozilla Common Voice (MCV), focusing on the Galician case within Proxecto N{\'o}s. We describe the organization of voice collection campaigns {--}on-site events, student participation, Validat{\'o}n marathons, and corporate collaboration{--} and analyze the results in MCV v22.0. While the dataset has achieved a modest scale, major gaps remain in metadata completeness and dialectal tagging, with implications for ASR performance. Drawing on our experience, we highlight effective strategies for engagement, such as transparent communication, cultural identification, and user-friendly tools. We conclude with lessons learnt for improving data representativeness, participant retention, and ethical governance. The observations are specific to the Galician case study but may inform similar efforts in other lesser-resourced languages."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="vladu-etal-2026-building">
<titleInfo>
<title>Building Collaborative Speech Corpora for Low-Resource Languages: The Galician Dataset in Mozilla Common Voice</title>
</titleInfo>
<name type="personal">
<namePart type="given">Adina</namePart>
<namePart type="given">Ioana</namePart>
<namePart type="family">Vladu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elisa</namePart>
<namePart type="family">Fernández Rei</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">María</namePart>
<namePart type="family">Pérez Lago</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper presents the methodology and outcomes of building collaborative speech corpora in Mozilla Common Voice (MCV), focusing on the Galician case within Proxecto Nós. We describe the organization of voice collection campaigns –on-site events, student participation, Validatón marathons, and corporate collaboration– and analyze the results in MCV v22.0. While the dataset has achieved a modest scale, major gaps remain in metadata completeness and dialectal tagging, with implications for ASR performance. Drawing on our experience, we highlight effective strategies for engagement, such as transparent communication, cultural identification, and user-friendly tools. We conclude with lessons learnt for improving data representativeness, participant retention, and ethical governance. The observations are specific to the Galician case study but may inform similar efforts in other lesser-resourced languages.</abstract>
<identifier type="citekey">vladu-etal-2026-building</identifier>
<identifier type="doi">10.63317/4gd79cq3cump</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.135/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>1710</start>
<end>1720</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Building Collaborative Speech Corpora for Low-Resource Languages: The Galician Dataset in Mozilla Common Voice
%A Vladu, Adina Ioana
%A Fernández Rei, Elisa
%A Pérez Lago, María
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F vladu-etal-2026-building
%X This paper presents the methodology and outcomes of building collaborative speech corpora in Mozilla Common Voice (MCV), focusing on the Galician case within Proxecto Nós. We describe the organization of voice collection campaigns –on-site events, student participation, Validatón marathons, and corporate collaboration– and analyze the results in MCV v22.0. While the dataset has achieved a modest scale, major gaps remain in metadata completeness and dialectal tagging, with implications for ASR performance. Drawing on our experience, we highlight effective strategies for engagement, such as transparent communication, cultural identification, and user-friendly tools. We conclude with lessons learnt for improving data representativeness, participant retention, and ethical governance. The observations are specific to the Galician case study but may inform similar efforts in other lesser-resourced languages.
%R 10.63317/4gd79cq3cump
%U https://aclanthology.org/2026.lrec-1.135/
%U https://doi.org/10.63317/4gd79cq3cump
%P 1710-1720
Markdown (Informal)
[Building Collaborative Speech Corpora for Low-Resource Languages: The Galician Dataset in Mozilla Common Voice](https://aclanthology.org/2026.lrec-1.135/) (Vladu et al., LREC 2026)
ACL