@inproceedings{bremang-etal-2026-creating,
title = "Creating Task-Specific Speech Recognition Datasets from Scratch for Low-Resource Languages: Assessing the Impact of Token Sequence Overlap",
author = "Bremang, Adwoa Asantewaa and
Asamoah Owusu, Dennis and
Quagraine, Victor Kow and
Annor-Adjaye, Leanne M.M.",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.240/",
doi = "10.63317/3myb33sgskfb",
pages = "3075--3082",
abstract = "Creating a task-specific speech recognition dataset is essential for developing speech recognition applications in low-resource languages. Such applications have uses in agriculture, finance, healthcare, and others, and benefit individuals with low literacy. However, a significant challenge is the high cost of data creation. While there is some work around cost-effective dataset selection, there is little to no work on building a cost-effective dataset for a task from scratch. Our work contributes to the latter. We created a speech recognition dataset from scratch and conducted two major sets of experiments. The first aimed to observe the effect of different datasets of the same size on model performance. Our results confirmed that the same amount spent collecting data can have vastly different results. The second experiment analyzed the effect of token sequence overlap between target and training data since a natural and intuitive approach to building a dataset from scratch for task would be having the task tokens occur in the training data. Our experiments showed that token sequence overlap was not the primary factor influencing model performance. Our work provides a counter-intuitive insight into building speech recognition datasets from scratch in low-resource settings and shows the need for further investigation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="bremang-etal-2026-creating">
<titleInfo>
<title>Creating Task-Specific Speech Recognition Datasets from Scratch for Low-Resource Languages: Assessing the Impact of Token Sequence Overlap</title>
</titleInfo>
<name type="personal">
<namePart type="given">Adwoa</namePart>
<namePart type="given">Asantewaa</namePart>
<namePart type="family">Bremang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dennis</namePart>
<namePart type="family">Asamoah Owusu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Victor</namePart>
<namePart type="given">Kow</namePart>
<namePart type="family">Quagraine</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Leanne</namePart>
<namePart type="given">M.M.</namePart>
<namePart type="family">Annor-Adjaye</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Creating a task-specific speech recognition dataset is essential for developing speech recognition applications in low-resource languages. Such applications have uses in agriculture, finance, healthcare, and others, and benefit individuals with low literacy. However, a significant challenge is the high cost of data creation. While there is some work around cost-effective dataset selection, there is little to no work on building a cost-effective dataset for a task from scratch. Our work contributes to the latter. We created a speech recognition dataset from scratch and conducted two major sets of experiments. The first aimed to observe the effect of different datasets of the same size on model performance. Our results confirmed that the same amount spent collecting data can have vastly different results. The second experiment analyzed the effect of token sequence overlap between target and training data since a natural and intuitive approach to building a dataset from scratch for task would be having the task tokens occur in the training data. Our experiments showed that token sequence overlap was not the primary factor influencing model performance. Our work provides a counter-intuitive insight into building speech recognition datasets from scratch in low-resource settings and shows the need for further investigation.</abstract>
<identifier type="citekey">bremang-etal-2026-creating</identifier>
<identifier type="doi">10.63317/3myb33sgskfb</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.240/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>3075</start>
<end>3082</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Creating Task-Specific Speech Recognition Datasets from Scratch for Low-Resource Languages: Assessing the Impact of Token Sequence Overlap
%A Bremang, Adwoa Asantewaa
%A Asamoah Owusu, Dennis
%A Quagraine, Victor Kow
%A Annor-Adjaye, Leanne M.M.
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F bremang-etal-2026-creating
%X Creating a task-specific speech recognition dataset is essential for developing speech recognition applications in low-resource languages. Such applications have uses in agriculture, finance, healthcare, and others, and benefit individuals with low literacy. However, a significant challenge is the high cost of data creation. While there is some work around cost-effective dataset selection, there is little to no work on building a cost-effective dataset for a task from scratch. Our work contributes to the latter. We created a speech recognition dataset from scratch and conducted two major sets of experiments. The first aimed to observe the effect of different datasets of the same size on model performance. Our results confirmed that the same amount spent collecting data can have vastly different results. The second experiment analyzed the effect of token sequence overlap between target and training data since a natural and intuitive approach to building a dataset from scratch for task would be having the task tokens occur in the training data. Our experiments showed that token sequence overlap was not the primary factor influencing model performance. Our work provides a counter-intuitive insight into building speech recognition datasets from scratch in low-resource settings and shows the need for further investigation.
%R 10.63317/3myb33sgskfb
%U https://aclanthology.org/2026.lrec-1.240/
%U https://doi.org/10.63317/3myb33sgskfb
%P 3075-3082
Markdown (Informal)
[Creating Task-Specific Speech Recognition Datasets from Scratch for Low-Resource Languages: Assessing the Impact of Token Sequence Overlap](https://aclanthology.org/2026.lrec-1.240/) (Bremang et al., LREC 2026)
ACL