@inproceedings{mohammadamini-etal-2026-exploring,
title = "Exploring the reusability of {N}orthern {K}urdish resources for Badini speech recognition",
author = "Mohammadamini, Mohammad and
Mohammed, Aveen Jalal and
Mohammed, Barzan Hussein and
Abdulazeez, Dezheen H. and
Sadeeq, Imad Saeed and
Salih, Dilgash Mohammed and
Melhum, Amera Ismail and
Dheyab, Abuobaida Abdullah",
editor = "Anastasopoulos, Antonis and
Markantonatou, Stella and
Ralli, Angela and
Zampieri, Marcos and
Bompolas, Stavros and
Stamou, Vivian",
booktitle = "Proceedings of the First Workshop on Dialects in {NLP} {---} A Resource Perspective",
month = may,
year = "2026",
address = "Palma de Mallorca",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.dialres-1.11/",
doi = "10.63317/2gzjngtqiqp7",
pages = "110--115",
abstract = "Badini is a variant of the Kurdish language spoken in the Duhok province of the Kurdistan Region of Iraq. It is written mainly in a modified version of the Arabic script. Although it shares the same script as Central Kurdish (CKB), it is linguistically classified under the Northern Kurdish (KMR) branch. In this paper, we explore the potential and limitations of Northern Kurdish ASR resources for the Badini variant. Firstly, we transliterate the Common Voice 18 dataset from the Latin script into the modified Arabic script and revised it to align with the orthographic conventions of Badini variant. Additionally, we introduce the first text collection for the Badini variant, containing 14,22 million tokens, which serves as a source for speech synthesis. A third resource developed in this research is a standard speech recognition benchmark recorded by 5 speakers which includes 2 hours and 46 minutes of multi-domain read speech. Results show that combining transliterated and synthetic data significantly improves recognition accuracy, achieving a 6.8{\%} CER and 34{\%} WER. All three resources curated during this research will be made available under the CC BY-NC-ND 4.0 license."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="mohammadamini-etal-2026-exploring">
<titleInfo>
<title>Exploring the reusability of Northern Kurdish resources for Badini speech recognition</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mohammad</namePart>
<namePart type="family">Mohammadamini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aveen</namePart>
<namePart type="given">Jalal</namePart>
<namePart type="family">Mohammed</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Barzan</namePart>
<namePart type="given">Hussein</namePart>
<namePart type="family">Mohammed</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dezheen</namePart>
<namePart type="given">H</namePart>
<namePart type="family">Abdulazeez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Imad</namePart>
<namePart type="given">Saeed</namePart>
<namePart type="family">Sadeeq</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dilgash</namePart>
<namePart type="given">Mohammed</namePart>
<namePart type="family">Salih</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Amera</namePart>
<namePart type="given">Ismail</namePart>
<namePart type="family">Melhum</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Abuobaida</namePart>
<namePart type="given">Abdullah</namePart>
<namePart type="family">Dheyab</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective</title>
</titleInfo>
<name type="personal">
<namePart type="given">Antonis</namePart>
<namePart type="family">Anastasopoulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stella</namePart>
<namePart type="family">Markantonatou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Angela</namePart>
<namePart type="family">Ralli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marcos</namePart>
<namePart type="family">Zampieri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stavros</namePart>
<namePart type="family">Bompolas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vivian</namePart>
<namePart type="family">Stamou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Badini is a variant of the Kurdish language spoken in the Duhok province of the Kurdistan Region of Iraq. It is written mainly in a modified version of the Arabic script. Although it shares the same script as Central Kurdish (CKB), it is linguistically classified under the Northern Kurdish (KMR) branch. In this paper, we explore the potential and limitations of Northern Kurdish ASR resources for the Badini variant. Firstly, we transliterate the Common Voice 18 dataset from the Latin script into the modified Arabic script and revised it to align with the orthographic conventions of Badini variant. Additionally, we introduce the first text collection for the Badini variant, containing 14,22 million tokens, which serves as a source for speech synthesis. A third resource developed in this research is a standard speech recognition benchmark recorded by 5 speakers which includes 2 hours and 46 minutes of multi-domain read speech. Results show that combining transliterated and synthetic data significantly improves recognition accuracy, achieving a 6.8% CER and 34% WER. All three resources curated during this research will be made available under the CC BY-NC-ND 4.0 license.</abstract>
<identifier type="citekey">mohammadamini-etal-2026-exploring</identifier>
<identifier type="doi">10.63317/2gzjngtqiqp7</identifier>
<location>
<url>https://aclanthology.org/2026.dialres-1.11/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>110</start>
<end>115</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Exploring the reusability of Northern Kurdish resources for Badini speech recognition
%A Mohammadamini, Mohammad
%A Mohammed, Aveen Jalal
%A Mohammed, Barzan Hussein
%A Abdulazeez, Dezheen H.
%A Sadeeq, Imad Saeed
%A Salih, Dilgash Mohammed
%A Melhum, Amera Ismail
%A Dheyab, Abuobaida Abdullah
%Y Anastasopoulos, Antonis
%Y Markantonatou, Stella
%Y Ralli, Angela
%Y Zampieri, Marcos
%Y Bompolas, Stavros
%Y Stamou, Vivian
%S Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca
%F mohammadamini-etal-2026-exploring
%X Badini is a variant of the Kurdish language spoken in the Duhok province of the Kurdistan Region of Iraq. It is written mainly in a modified version of the Arabic script. Although it shares the same script as Central Kurdish (CKB), it is linguistically classified under the Northern Kurdish (KMR) branch. In this paper, we explore the potential and limitations of Northern Kurdish ASR resources for the Badini variant. Firstly, we transliterate the Common Voice 18 dataset from the Latin script into the modified Arabic script and revised it to align with the orthographic conventions of Badini variant. Additionally, we introduce the first text collection for the Badini variant, containing 14,22 million tokens, which serves as a source for speech synthesis. A third resource developed in this research is a standard speech recognition benchmark recorded by 5 speakers which includes 2 hours and 46 minutes of multi-domain read speech. Results show that combining transliterated and synthetic data significantly improves recognition accuracy, achieving a 6.8% CER and 34% WER. All three resources curated during this research will be made available under the CC BY-NC-ND 4.0 license.
%R 10.63317/2gzjngtqiqp7
%U https://aclanthology.org/2026.dialres-1.11/
%U https://doi.org/10.63317/2gzjngtqiqp7
%P 110-115
Markdown (Informal)
[Exploring the reusability of Northern Kurdish resources for Badini speech recognition](https://aclanthology.org/2026.dialres-1.11/) (Mohammadamini et al., DialRes 2026)
ACL
- Mohammad Mohammadamini, Aveen Jalal Mohammed, Barzan Hussein Mohammed, Dezheen H. Abdulazeez, Imad Saeed Sadeeq, Dilgash Mohammed Salih, Amera Ismail Melhum, and Abuobaida Abdullah Dheyab. 2026. Exploring the reusability of Northern Kurdish resources for Badini speech recognition. In Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective, pages 110–115, Palma de Mallorca. Association for Computational Linguistics.