@inproceedings{knight-alva-manchego-2026-corpus,
title = "From Corpus to Community: New {NLP} Tools for {W}elsh Language Research and Learning",
author = "Knight, Dawn and
Alva-Manchego, Fernando",
editor = "Ba{\'n}ski, Piotr and
Knight, Dawn and
Kupietz, Marc and
Witt, Andreas and
Wr{\'o}blewska, Alina",
booktitle = "Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cmlc-1.17/",
doi = "10.63317/56urzbii2rvy",
pages = "101--103",
abstract = "Launched in 2020, CorCenCC (Corpws Cenedlaethol Cymraeg Cyfoes {--} National Corpus of Contemporary Welsh) is the first large-scale corpus of the Welsh language to integrate spoken, written, and electronically mediated data, offering a comprehensive snapshot of contemporary Welsh use. Including contributions from over 2,000 speakers, the 11.2-million-word corpus represents the diversity of Wales{'}s linguistic landscape. As a national resource, CorCenCC enables users to explore real world Welsh. Several tools and resources were developed through the CorCenCC project, including the CyTag POS tagger and CySemTag (adapted from Lancaster University{'}s USAS semantic system), to enable the grammatical and semantic categorisation of the dataset. The team also built the pedagogic toolkit Y Tiwtiadur, to allow learners and teachers to access corpus-based examples and tasks. Additionally, Yr Amliadur provides curated frequency-based wordlists across modes and parts of speech, supporting linguistic analysis and vocabulary development. Since completing the corpus, the team has focused on extending its impact and reach, to ensure that the resources are maintained and sustained for future use; a challenge often faced when large-scale projects end. This poster profiles the tools and resources created from and inspired by CorCenCC and its associated tools and resources, as a means of supporting the democratisation of linguistic resources for minoritised language contexts."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="knight-alva-manchego-2026-corpus">
<titleInfo>
<title>From Corpus to Community: New NLP Tools for Welsh Language Research and Learning</title>
</titleInfo>
<name type="personal">
<namePart type="given">Dawn</namePart>
<namePart type="family">Knight</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fernando</namePart>
<namePart type="family">Alva-Manchego</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Piotr</namePart>
<namePart type="family">Bański</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dawn</namePart>
<namePart type="family">Knight</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marc</namePart>
<namePart type="family">Kupietz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andreas</namePart>
<namePart type="family">Witt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Wróblewska</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Launched in 2020, CorCenCC (Corpws Cenedlaethol Cymraeg Cyfoes – National Corpus of Contemporary Welsh) is the first large-scale corpus of the Welsh language to integrate spoken, written, and electronically mediated data, offering a comprehensive snapshot of contemporary Welsh use. Including contributions from over 2,000 speakers, the 11.2-million-word corpus represents the diversity of Wales’s linguistic landscape. As a national resource, CorCenCC enables users to explore real world Welsh. Several tools and resources were developed through the CorCenCC project, including the CyTag POS tagger and CySemTag (adapted from Lancaster University’s USAS semantic system), to enable the grammatical and semantic categorisation of the dataset. The team also built the pedagogic toolkit Y Tiwtiadur, to allow learners and teachers to access corpus-based examples and tasks. Additionally, Yr Amliadur provides curated frequency-based wordlists across modes and parts of speech, supporting linguistic analysis and vocabulary development. Since completing the corpus, the team has focused on extending its impact and reach, to ensure that the resources are maintained and sustained for future use; a challenge often faced when large-scale projects end. This poster profiles the tools and resources created from and inspired by CorCenCC and its associated tools and resources, as a means of supporting the democratisation of linguistic resources for minoritised language contexts.</abstract>
<identifier type="citekey">knight-alva-manchego-2026-corpus</identifier>
<identifier type="doi">10.63317/56urzbii2rvy</identifier>
<location>
<url>https://aclanthology.org/2026.cmlc-1.17/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>101</start>
<end>103</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T From Corpus to Community: New NLP Tools for Welsh Language Research and Learning
%A Knight, Dawn
%A Alva-Manchego, Fernando
%Y Bański, Piotr
%Y Knight, Dawn
%Y Kupietz, Marc
%Y Witt, Andreas
%Y Wróblewska, Alina
%S Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F knight-alva-manchego-2026-corpus
%X Launched in 2020, CorCenCC (Corpws Cenedlaethol Cymraeg Cyfoes – National Corpus of Contemporary Welsh) is the first large-scale corpus of the Welsh language to integrate spoken, written, and electronically mediated data, offering a comprehensive snapshot of contemporary Welsh use. Including contributions from over 2,000 speakers, the 11.2-million-word corpus represents the diversity of Wales’s linguistic landscape. As a national resource, CorCenCC enables users to explore real world Welsh. Several tools and resources were developed through the CorCenCC project, including the CyTag POS tagger and CySemTag (adapted from Lancaster University’s USAS semantic system), to enable the grammatical and semantic categorisation of the dataset. The team also built the pedagogic toolkit Y Tiwtiadur, to allow learners and teachers to access corpus-based examples and tasks. Additionally, Yr Amliadur provides curated frequency-based wordlists across modes and parts of speech, supporting linguistic analysis and vocabulary development. Since completing the corpus, the team has focused on extending its impact and reach, to ensure that the resources are maintained and sustained for future use; a challenge often faced when large-scale projects end. This poster profiles the tools and resources created from and inspired by CorCenCC and its associated tools and resources, as a means of supporting the democratisation of linguistic resources for minoritised language contexts.
%R 10.63317/56urzbii2rvy
%U https://aclanthology.org/2026.cmlc-1.17/
%U https://doi.org/10.63317/56urzbii2rvy
%P 101-103
Markdown (Informal)
[From Corpus to Community: New NLP Tools for Welsh Language Research and Learning](https://aclanthology.org/2026.cmlc-1.17/) (Knight & Alva-Manchego, CMLC 2026)
ACL