@inproceedings{walsh-etal-2026-quality,
title = "Quality and Appropriateness of Large Text Datasets for {I}rish {NLP}",
author = "Walsh, Abigail and
Andrade, Mark and
Adkins, Jane Lauren and
O{'}Connell, Ornait and
O{'}Connor, {\'E}anna and
Rushe, Ellen and
Davis, Brian",
editor = "Ojha, Atul Kr. and
Sakti, Sakriani and
Soria, Claudia and
Melero, Maite and
McCrae, John P. and
Lignos, Constantine and
Liu, Chao-Hong and
Claramunt, German Rigau and
Rehm, Georg",
booktitle = "Proceedings of the {SIGUL} 2026 Joint Workshop with {ELE}, {EURALI}, and {DCLRL}: Towards Inclusivity and Equality: Language Resources and Technologies for Under-Resourced and Endangered Languages",
month = may,
year = "2026",
address = "Palma, Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.sigul-1.14/",
doi = "10.63317/3sxe9j64u492",
pages = "126--142",
abstract = "The value of high-quality datasets for training essential language tools has long been recognised for NLP research. Despite the importance of such datasets, most language data available for training consists of large, automatically curated corpora, often scraped from web content. The quality of such datasets is often an unknown factor. This presents a problem for already low-resourced languages (such as Irish), as existing datasets may not provide adequate, representative language data for training effective models. This paper examines existing monolingual and parallel Irish text corpora to evaluate the quality of the language data, through manual review, automatic metrics, and LLMs as judges."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="walsh-etal-2026-quality">
<titleInfo>
<title>Quality and Appropriateness of Large Text Datasets for Irish NLP</title>
</titleInfo>
<name type="personal">
<namePart type="given">Abigail</namePart>
<namePart type="family">Walsh</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mark</namePart>
<namePart type="family">Andrade</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jane</namePart>
<namePart type="given">Lauren</namePart>
<namePart type="family">Adkins</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ornait</namePart>
<namePart type="family">O’Connell</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Éanna</namePart>
<namePart type="family">O’Connor</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ellen</namePart>
<namePart type="family">Rushe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Brian</namePart>
<namePart type="family">Davis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the SIGUL 2026 Joint Workshop with ELE, EURALI, and DCLRL: Towards Inclusivity and Equality: Language Resources and Technologies for Under-Resourced and Endangered Languages</title>
</titleInfo>
<name type="personal">
<namePart type="given">Atul</namePart>
<namePart type="given">Kr.</namePart>
<namePart type="family">Ojha</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sakriani</namePart>
<namePart type="family">Sakti</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Claudia</namePart>
<namePart type="family">Soria</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maite</namePart>
<namePart type="family">Melero</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">John</namePart>
<namePart type="given">P</namePart>
<namePart type="family">McCrae</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Constantine</namePart>
<namePart type="family">Lignos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chao-Hong</namePart>
<namePart type="family">Liu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">German</namePart>
<namePart type="given">Rigau</namePart>
<namePart type="family">Claramunt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Georg</namePart>
<namePart type="family">Rehm</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The value of high-quality datasets for training essential language tools has long been recognised for NLP research. Despite the importance of such datasets, most language data available for training consists of large, automatically curated corpora, often scraped from web content. The quality of such datasets is often an unknown factor. This presents a problem for already low-resourced languages (such as Irish), as existing datasets may not provide adequate, representative language data for training effective models. This paper examines existing monolingual and parallel Irish text corpora to evaluate the quality of the language data, through manual review, automatic metrics, and LLMs as judges.</abstract>
<identifier type="citekey">walsh-etal-2026-quality</identifier>
<identifier type="doi">10.63317/3sxe9j64u492</identifier>
<location>
<url>https://aclanthology.org/2026.sigul-1.14/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>126</start>
<end>142</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Quality and Appropriateness of Large Text Datasets for Irish NLP
%A Walsh, Abigail
%A Andrade, Mark
%A Adkins, Jane Lauren
%A O’Connell, Ornait
%A O’Connor, Éanna
%A Rushe, Ellen
%A Davis, Brian
%Y Ojha, Atul Kr.
%Y Sakti, Sakriani
%Y Soria, Claudia
%Y Melero, Maite
%Y McCrae, John P.
%Y Lignos, Constantine
%Y Liu, Chao-Hong
%Y Claramunt, German Rigau
%Y Rehm, Georg
%S Proceedings of the SIGUL 2026 Joint Workshop with ELE, EURALI, and DCLRL: Towards Inclusivity and Equality: Language Resources and Technologies for Under-Resourced and Endangered Languages
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca, Spain
%F walsh-etal-2026-quality
%X The value of high-quality datasets for training essential language tools has long been recognised for NLP research. Despite the importance of such datasets, most language data available for training consists of large, automatically curated corpora, often scraped from web content. The quality of such datasets is often an unknown factor. This presents a problem for already low-resourced languages (such as Irish), as existing datasets may not provide adequate, representative language data for training effective models. This paper examines existing monolingual and parallel Irish text corpora to evaluate the quality of the language data, through manual review, automatic metrics, and LLMs as judges.
%R 10.63317/3sxe9j64u492
%U https://aclanthology.org/2026.sigul-1.14/
%U https://doi.org/10.63317/3sxe9j64u492
%P 126-142
Markdown (Informal)
[Quality and Appropriateness of Large Text Datasets for Irish NLP](https://aclanthology.org/2026.sigul-1.14/) (Walsh et al., SIGUL-EURALI-DCLRL 2026)
ACL
- Abigail Walsh, Mark Andrade, Jane Lauren Adkins, Ornait O’Connell, Éanna O’Connor, Ellen Rushe, and Brian Davis. 2026. Quality and Appropriateness of Large Text Datasets for Irish NLP. In Proceedings of the SIGUL 2026 Joint Workshop with ELE, EURALI, and DCLRL: Towards Inclusivity and Equality: Language Resources and Technologies for Under-Resourced and Endangered Languages, pages 126–142, Palma, Mallorca, Spain. ELRA Language Resources Association (ELRA).