@article{sedlacek-etal-2026-orca,
title = "{ORCA}: Open-ended Response Correctness Assessment for Audio Question Answering",
author = "Sedl{\'a}{\v{c}}ek, {\v{S}}imon and
Barahona, Sara and
Bola{\~n}os, Cecilia and
Herrera-Alarc{\'o}n, Laura and
Udupa, Sathvik and
L{\'o}pez, Fernando and
Ferner, Allison and
Yusuf, Bolaji and
Lozano-Diez, Alicia and
Kesiraju, Santosh and
Duraiswami, Ramani and
{\v{C}}ernock{\'y}, Jan",
journal = "Transactions of the Association for Computational Linguistics",
volume = "14",
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.tacl-1.100/",
doi = "10.1162/tacl.a.798",
pages = "2213--2233",
abstract = "Reliable assessment of the abilities of large audio language models (LALMs) is essential to advancing the state of the art. As benchmarks rapidly evolve to incorporate complex reasoning and subjective tasks, they increasingly necessitate open-ended responses from LALMs. We present Open-ended Response Correctness Assessment (ORCA){---}a reliable and lightweight model-based approach for answer correctness and disagreement modeling. We employ a three-stage annotation pipeline combining human judgment, structured feedback, and human-AI correction, yielding 9,663 annotations across 3,699 question-answer pairs from 15 LALMs on three audio understanding and reasoning benchmarks (achieving a Krippendorff{'}s alpha of 0.82). Our experiments employing curriculum learning show that ORCA models achieve a Spearman correlation of 0.91 with average human correctness ratings on seen benchmarks and generalize to unseen benchmarks with a score of 0.85, outperforming several LLM judge baselines including Gemini 2.5 Flash. Furthermore, we demonstrate that ORCA{'}s predicted variance correlates strongly with human disagreement, allowing it to effectively identify problematic benchmark items."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="sedlacek-etal-2026-orca">
<titleInfo>
<title>ORCA: Open-ended Response Correctness Assessment for Audio Question Answering</title>
</titleInfo>
<name type="personal">
<namePart type="given">Šimon</namePart>
<namePart type="family">Sedláček</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sara</namePart>
<namePart type="family">Barahona</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Cecilia</namePart>
<namePart type="family">Bolaños</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Laura</namePart>
<namePart type="family">Herrera-Alarcón</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sathvik</namePart>
<namePart type="family">Udupa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fernando</namePart>
<namePart type="family">López</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Allison</namePart>
<namePart type="family">Ferner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Bolaji</namePart>
<namePart type="family">Yusuf</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alicia</namePart>
<namePart type="family">Lozano-Diez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Santosh</namePart>
<namePart type="family">Kesiraju</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ramani</namePart>
<namePart type="family">Duraiswami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jan</namePart>
<namePart type="family">Černocký</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>Reliable assessment of the abilities of large audio language models (LALMs) is essential to advancing the state of the art. As benchmarks rapidly evolve to incorporate complex reasoning and subjective tasks, they increasingly necessitate open-ended responses from LALMs. We present Open-ended Response Correctness Assessment (ORCA)—a reliable and lightweight model-based approach for answer correctness and disagreement modeling. We employ a three-stage annotation pipeline combining human judgment, structured feedback, and human-AI correction, yielding 9,663 annotations across 3,699 question-answer pairs from 15 LALMs on three audio understanding and reasoning benchmarks (achieving a Krippendorff’s alpha of 0.82). Our experiments employing curriculum learning show that ORCA models achieve a Spearman correlation of 0.91 with average human correctness ratings on seen benchmarks and generalize to unseen benchmarks with a score of 0.85, outperforming several LLM judge baselines including Gemini 2.5 Flash. Furthermore, we demonstrate that ORCA’s predicted variance correlates strongly with human disagreement, allowing it to effectively identify problematic benchmark items.</abstract>
<identifier type="citekey">sedlacek-etal-2026-orca</identifier>
<identifier type="doi">10.1162/tacl.a.798</identifier>
<location>
<url>https://aclanthology.org/2026.tacl-1.100/</url>
</location>
<part>
<date>2026</date>
<detail type="volume"><number>14</number></detail>
<extent unit="page">
<start>2213</start>
<end>2233</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T ORCA: Open-ended Response Correctness Assessment for Audio Question Answering
%A Sedláček, Šimon
%A Barahona, Sara
%A Bolaños, Cecilia
%A Herrera-Alarcón, Laura
%A Udupa, Sathvik
%A López, Fernando
%A Ferner, Allison
%A Yusuf, Bolaji
%A Lozano-Diez, Alicia
%A Kesiraju, Santosh
%A Duraiswami, Ramani
%A Černocký, Jan
%J Transactions of the Association for Computational Linguistics
%D 2026
%V 14
%I MIT Press
%C Cambridge, MA
%F sedlacek-etal-2026-orca
%X Reliable assessment of the abilities of large audio language models (LALMs) is essential to advancing the state of the art. As benchmarks rapidly evolve to incorporate complex reasoning and subjective tasks, they increasingly necessitate open-ended responses from LALMs. We present Open-ended Response Correctness Assessment (ORCA)—a reliable and lightweight model-based approach for answer correctness and disagreement modeling. We employ a three-stage annotation pipeline combining human judgment, structured feedback, and human-AI correction, yielding 9,663 annotations across 3,699 question-answer pairs from 15 LALMs on three audio understanding and reasoning benchmarks (achieving a Krippendorff’s alpha of 0.82). Our experiments employing curriculum learning show that ORCA models achieve a Spearman correlation of 0.91 with average human correctness ratings on seen benchmarks and generalize to unseen benchmarks with a score of 0.85, outperforming several LLM judge baselines including Gemini 2.5 Flash. Furthermore, we demonstrate that ORCA’s predicted variance correlates strongly with human disagreement, allowing it to effectively identify problematic benchmark items.
%R 10.1162/tacl.a.798
%U https://aclanthology.org/2026.tacl-1.100/
%U https://doi.org/10.1162/tacl.a.798
%P 2213-2233
Markdown (Informal)
[ORCA: Open-ended Response Correctness Assessment for Audio Question Answering](https://aclanthology.org/2026.tacl-1.100/) (Sedláček et al., TACL 2026)
ACL
- Šimon Sedláček, Sara Barahona, Cecilia Bolaños, Laura Herrera-Alarcón, Sathvik Udupa, Fernando López, Allison Ferner, Bolaji Yusuf, Alicia Lozano-Diez, Santosh Kesiraju, Ramani Duraiswami, and Jan Černocký. 2026. ORCA: Open-ended Response Correctness Assessment for Audio Question Answering. Transactions of the Association for Computational Linguistics, 14:2213–2233.