@inproceedings{al-khalifa-2026-omniasr,
title = "When Does {O}mni{ASR} Fail? A Fine-Grained Human Evaluation on Saudi {A}rabic Dialects",
author = "Al-Khalifa, Hend",
editor = "Hosseini-Kivanani, Nina and
Brutti, Alessio and
Matassoni, Marco and
Dowerah, Sandipana and
Liga, Davide and
Schommer, Christoph",
booktitle = "Proceedings of Speech Language Models in Low-Resource Settings: Performance, Evaluation, and Bias Analysis ({SPEAKABLE}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.speakable-1.13/",
doi = "10.63317/2wmxhnwmucq4",
pages = "118--124",
abstract = "Automatic Speech Recognition (ASR) evaluation has traditionally relied on Word Error Rate (WER), a metric that treats all errors equally and obscures critical failure modes. In this paper, we present a fine-grained human evaluation of Meta{'}s recently released OmniASR system on Saudi Arabic dialects using the SADA dataset. Three trained annotators evaluated 103 audio samples, producing 264 annotations across two dimensions (comprehensibility and naturalness) while categorizing errors using a novel 10-category Arabic-specific error taxonomy. OmniASR achieved a mean WER of 42.2{\%} and mean comprehensibility of 3.62/5, but exhibited a bimodal performance pattern: 32.6{\%} of transcriptions achieved perfect scores while 21.2{\%} were essentially unusable. Error analysis reveals that hallucinations and deletions have the greatest negative impact on comprehensibility ({\ensuremath{-}}1.64 and {\ensuremath{-}}1.57 points respectively), roughly 6{\texttimes} more damaging than named entity errors. Importantly, WER correlates only moderately with human comprehensibility ratings (r = {\ensuremath{-}}0.679), explaining just 46{\%} of variance in human judgments. These findings demonstrate the limitations of WER as a sole evaluation metric and highlight the need for human-centered, error-type-aware evaluation frameworks for Arabic ASR systems."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="al-khalifa-2026-omniasr">
<titleInfo>
<title>When Does OmniASR Fail? A Fine-Grained Human Evaluation on Saudi Arabic Dialects</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hend</namePart>
<namePart type="family">Al-Khalifa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Speech Language Models in Low-Resource Settings: Performance, Evaluation, and Bias Analysis (SPEAKABLE) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Nina</namePart>
<namePart type="family">Hosseini-Kivanani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alessio</namePart>
<namePart type="family">Brutti</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marco</namePart>
<namePart type="family">Matassoni</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sandipana</namePart>
<namePart type="family">Dowerah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Davide</namePart>
<namePart type="family">Liga</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christoph</namePart>
<namePart type="family">Schommer</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Automatic Speech Recognition (ASR) evaluation has traditionally relied on Word Error Rate (WER), a metric that treats all errors equally and obscures critical failure modes. In this paper, we present a fine-grained human evaluation of Meta’s recently released OmniASR system on Saudi Arabic dialects using the SADA dataset. Three trained annotators evaluated 103 audio samples, producing 264 annotations across two dimensions (comprehensibility and naturalness) while categorizing errors using a novel 10-category Arabic-specific error taxonomy. OmniASR achieved a mean WER of 42.2% and mean comprehensibility of 3.62/5, but exhibited a bimodal performance pattern: 32.6% of transcriptions achieved perfect scores while 21.2% were essentially unusable. Error analysis reveals that hallucinations and deletions have the greatest negative impact on comprehensibility (\ensuremath-1.64 and \ensuremath-1.57 points respectively), roughly 6× more damaging than named entity errors. Importantly, WER correlates only moderately with human comprehensibility ratings (r = \ensuremath-0.679), explaining just 46% of variance in human judgments. These findings demonstrate the limitations of WER as a sole evaluation metric and highlight the need for human-centered, error-type-aware evaluation frameworks for Arabic ASR systems.</abstract>
<identifier type="citekey">al-khalifa-2026-omniasr</identifier>
<identifier type="doi">10.63317/2wmxhnwmucq4</identifier>
<location>
<url>https://aclanthology.org/2026.speakable-1.13/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>118</start>
<end>124</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T When Does OmniASR Fail? A Fine-Grained Human Evaluation on Saudi Arabic Dialects
%A Al-Khalifa, Hend
%Y Hosseini-Kivanani, Nina
%Y Brutti, Alessio
%Y Matassoni, Marco
%Y Dowerah, Sandipana
%Y Liga, Davide
%Y Schommer, Christoph
%S Proceedings of Speech Language Models in Low-Resource Settings: Performance, Evaluation, and Bias Analysis (SPEAKABLE) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F al-khalifa-2026-omniasr
%X Automatic Speech Recognition (ASR) evaluation has traditionally relied on Word Error Rate (WER), a metric that treats all errors equally and obscures critical failure modes. In this paper, we present a fine-grained human evaluation of Meta’s recently released OmniASR system on Saudi Arabic dialects using the SADA dataset. Three trained annotators evaluated 103 audio samples, producing 264 annotations across two dimensions (comprehensibility and naturalness) while categorizing errors using a novel 10-category Arabic-specific error taxonomy. OmniASR achieved a mean WER of 42.2% and mean comprehensibility of 3.62/5, but exhibited a bimodal performance pattern: 32.6% of transcriptions achieved perfect scores while 21.2% were essentially unusable. Error analysis reveals that hallucinations and deletions have the greatest negative impact on comprehensibility (\ensuremath-1.64 and \ensuremath-1.57 points respectively), roughly 6× more damaging than named entity errors. Importantly, WER correlates only moderately with human comprehensibility ratings (r = \ensuremath-0.679), explaining just 46% of variance in human judgments. These findings demonstrate the limitations of WER as a sole evaluation metric and highlight the need for human-centered, error-type-aware evaluation frameworks for Arabic ASR systems.
%R 10.63317/2wmxhnwmucq4
%U https://aclanthology.org/2026.speakable-1.13/
%U https://doi.org/10.63317/2wmxhnwmucq4
%P 118-124
Markdown (Informal)
[When Does OmniASR Fail? A Fine-Grained Human Evaluation on Saudi Arabic Dialects](https://aclanthology.org/2026.speakable-1.13/) (Al-Khalifa, SPEAKABLE 2026)
ACL