@inproceedings{alamr-etal-2026-thaka,
title = "Thaka at {KSAA}-2026 Task 2: Regularized Fine-Tuning for {A}rabic Speech Diacritization",
author = "Alamr, Meshal Abdullah and
Alqaeri, Hassan Rshed and
Aldahlawi, Abdullah",
editor = "Al-Khalifa, Hend and
El-Haj, Mo and
Ezzini, Saad",
booktitle = "The 7th Workshop on Open-Source {A}rabic Corpora and Processing Tools ({OSACT}7) with 5 Shared Tasks",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.osact-1.29/",
doi = "10.63317/4iat33d2ge52",
pages = "225--228",
abstract = "We describe the winning system for Task 2 of the KSAA-2026 Shared Task on Arabic Speech Dictation with Automatic Diacritization. The task requires producing fully diacritized Arabic text from speech audio and undiacritized transcripts, with only 2,327 training samples available and no external data permitted. Our system fine-tunes CATT-Whisper, a character-level multimodal model combining a pretrained CATT text encoder with a frozen Whisper speech encoder. The key to our approach is training regularization: R-Drop consistency regularization, Optuna-optimized hyperparameters with high weight decay, and Focal Loss. At inference, we average 200 stochastic forward passes across four model checkpoints using Monte Carlo Dropout at the softmax probability level. The system achieves 23.26{\%} WER on the primary leaderboard metric (with case endings, including no-diacritic positions), placing 1st among all participants."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="alamr-etal-2026-thaka">
<titleInfo>
<title>Thaka at KSAA-2026 Task 2: Regularized Fine-Tuning for Arabic Speech Diacritization</title>
</titleInfo>
<name type="personal">
<namePart type="given">Meshal</namePart>
<namePart type="given">Abdullah</namePart>
<namePart type="family">Alamr</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hassan</namePart>
<namePart type="given">Rshed</namePart>
<namePart type="family">Alqaeri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Abdullah</namePart>
<namePart type="family">Aldahlawi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hend</namePart>
<namePart type="family">Al-Khalifa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We describe the winning system for Task 2 of the KSAA-2026 Shared Task on Arabic Speech Dictation with Automatic Diacritization. The task requires producing fully diacritized Arabic text from speech audio and undiacritized transcripts, with only 2,327 training samples available and no external data permitted. Our system fine-tunes CATT-Whisper, a character-level multimodal model combining a pretrained CATT text encoder with a frozen Whisper speech encoder. The key to our approach is training regularization: R-Drop consistency regularization, Optuna-optimized hyperparameters with high weight decay, and Focal Loss. At inference, we average 200 stochastic forward passes across four model checkpoints using Monte Carlo Dropout at the softmax probability level. The system achieves 23.26% WER on the primary leaderboard metric (with case endings, including no-diacritic positions), placing 1st among all participants.</abstract>
<identifier type="citekey">alamr-etal-2026-thaka</identifier>
<identifier type="doi">10.63317/4iat33d2ge52</identifier>
<location>
<url>https://aclanthology.org/2026.osact-1.29/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>225</start>
<end>228</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Thaka at KSAA-2026 Task 2: Regularized Fine-Tuning for Arabic Speech Diacritization
%A Alamr, Meshal Abdullah
%A Alqaeri, Hassan Rshed
%A Aldahlawi, Abdullah
%Y Al-Khalifa, Hend
%Y El-Haj, Mo
%Y Ezzini, Saad
%S The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma, Mallorca (Spain)
%F alamr-etal-2026-thaka
%X We describe the winning system for Task 2 of the KSAA-2026 Shared Task on Arabic Speech Dictation with Automatic Diacritization. The task requires producing fully diacritized Arabic text from speech audio and undiacritized transcripts, with only 2,327 training samples available and no external data permitted. Our system fine-tunes CATT-Whisper, a character-level multimodal model combining a pretrained CATT text encoder with a frozen Whisper speech encoder. The key to our approach is training regularization: R-Drop consistency regularization, Optuna-optimized hyperparameters with high weight decay, and Focal Loss. At inference, we average 200 stochastic forward passes across four model checkpoints using Monte Carlo Dropout at the softmax probability level. The system achieves 23.26% WER on the primary leaderboard metric (with case endings, including no-diacritic positions), placing 1st among all participants.
%R 10.63317/4iat33d2ge52
%U https://aclanthology.org/2026.osact-1.29/
%U https://doi.org/10.63317/4iat33d2ge52
%P 225-228
Markdown (Informal)
[Thaka at KSAA-2026 Task 2: Regularized Fine-Tuning for Arabic Speech Diacritization](https://aclanthology.org/2026.osact-1.29/) (Alamr et al., OSACT 2026)
ACL