@inproceedings{alharthi-etal-2026-abjad,
title = "Abjad {AI} at {KSAA}-2026 Shared Task 2: Grouped Speech Conditioning for {A}rabic Diacritization",
author = "Alharthi, Naif Saad and
Ghannam, Ahmad and
Alasmary, Faris and
Al Tabash, Kholood and
Sadah, Shouq and
Ghouti, Lahouari",
editor = "Al-Khalifa, Hend and
El-Haj, Mo and
Ezzini, Saad",
booktitle = "The 7th Workshop on Open-Source {A}rabic Corpora and Processing Tools ({OSACT}7) with 5 Shared Tasks",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.osact-1.33/",
doi = "10.63317/4t229uhjkkot",
pages = "252--255",
abstract = "We describe Abjad AI{'}s submission to KSAA-2026 Shared Task 2 on automatic diacritization of Arabic speech dictation. The task requires generating fully diacritized text given speech audio and an undiacritized transcript. Because text-only diacritization cannot resolve ambiguities that are recoverable from the acoustic signal, we propose conditioning a character-level encoder-only Transformer (CATT) (Alasmary et al., 2024) on speech representations. We introduce grouped speech conditioning, which downsamples speech encoder features into a small set of pooled tokens concatenated to the text input, enabling efficient fusion without architectural changes to CATT. We train with a two-phase schedule that first freezes the text encoder, then fine-tunes the full model. Our best system, using Whisper-small (Rad-ford et al., 2022) features with five grouped tokens, achieves a Diacritization Error Rate (DER) of 6.60 and a Word Error Rate (WER) of 18.66 (without case endings, including no-diacritic) on the official test set. Notably, we find that Whisper-small consistently outperforms Whisper-large-v3, suggesting that compact speech representations better suit this fusion setting. This is an extended and revised version of our previous work (Ghannam et al., 2025)."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="alharthi-etal-2026-abjad">
<titleInfo>
<title>Abjad AI at KSAA-2026 Shared Task 2: Grouped Speech Conditioning for Arabic Diacritization</title>
</titleInfo>
<name type="personal">
<namePart type="given">Naif</namePart>
<namePart type="given">Saad</namePart>
<namePart type="family">Alharthi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ahmad</namePart>
<namePart type="family">Ghannam</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Faris</namePart>
<namePart type="family">Alasmary</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kholood</namePart>
<namePart type="family">Al Tabash</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shouq</namePart>
<namePart type="family">Sadah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lahouari</namePart>
<namePart type="family">Ghouti</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hend</namePart>
<namePart type="family">Al-Khalifa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We describe Abjad AI’s submission to KSAA-2026 Shared Task 2 on automatic diacritization of Arabic speech dictation. The task requires generating fully diacritized text given speech audio and an undiacritized transcript. Because text-only diacritization cannot resolve ambiguities that are recoverable from the acoustic signal, we propose conditioning a character-level encoder-only Transformer (CATT) (Alasmary et al., 2024) on speech representations. We introduce grouped speech conditioning, which downsamples speech encoder features into a small set of pooled tokens concatenated to the text input, enabling efficient fusion without architectural changes to CATT. We train with a two-phase schedule that first freezes the text encoder, then fine-tunes the full model. Our best system, using Whisper-small (Rad-ford et al., 2022) features with five grouped tokens, achieves a Diacritization Error Rate (DER) of 6.60 and a Word Error Rate (WER) of 18.66 (without case endings, including no-diacritic) on the official test set. Notably, we find that Whisper-small consistently outperforms Whisper-large-v3, suggesting that compact speech representations better suit this fusion setting. This is an extended and revised version of our previous work (Ghannam et al., 2025).</abstract>
<identifier type="citekey">alharthi-etal-2026-abjad</identifier>
<identifier type="doi">10.63317/4t229uhjkkot</identifier>
<location>
<url>https://aclanthology.org/2026.osact-1.33/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>252</start>
<end>255</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Abjad AI at KSAA-2026 Shared Task 2: Grouped Speech Conditioning for Arabic Diacritization
%A Alharthi, Naif Saad
%A Ghannam, Ahmad
%A Alasmary, Faris
%A Al Tabash, Kholood
%A Sadah, Shouq
%A Ghouti, Lahouari
%Y Al-Khalifa, Hend
%Y El-Haj, Mo
%Y Ezzini, Saad
%S The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma, Mallorca (Spain)
%F alharthi-etal-2026-abjad
%X We describe Abjad AI’s submission to KSAA-2026 Shared Task 2 on automatic diacritization of Arabic speech dictation. The task requires generating fully diacritized text given speech audio and an undiacritized transcript. Because text-only diacritization cannot resolve ambiguities that are recoverable from the acoustic signal, we propose conditioning a character-level encoder-only Transformer (CATT) (Alasmary et al., 2024) on speech representations. We introduce grouped speech conditioning, which downsamples speech encoder features into a small set of pooled tokens concatenated to the text input, enabling efficient fusion without architectural changes to CATT. We train with a two-phase schedule that first freezes the text encoder, then fine-tunes the full model. Our best system, using Whisper-small (Rad-ford et al., 2022) features with five grouped tokens, achieves a Diacritization Error Rate (DER) of 6.60 and a Word Error Rate (WER) of 18.66 (without case endings, including no-diacritic) on the official test set. Notably, we find that Whisper-small consistently outperforms Whisper-large-v3, suggesting that compact speech representations better suit this fusion setting. This is an extended and revised version of our previous work (Ghannam et al., 2025).
%R 10.63317/4t229uhjkkot
%U https://aclanthology.org/2026.osact-1.33/
%U https://doi.org/10.63317/4t229uhjkkot
%P 252-255
Markdown (Informal)
[Abjad AI at KSAA-2026 Shared Task 2: Grouped Speech Conditioning for Arabic Diacritization](https://aclanthology.org/2026.osact-1.33/) (Alharthi et al., OSACT 2026)
ACL