@inproceedings{iakivchuk-2026-using,
title = "Using Model Disagreement to Identify Unstable Regions in {MT} Evaluation",
author = "Iakivchuk, Vitalii",
editor = "Shterionov, Dimitar and
Vanmassenhove, Eva and
De Sisto, Mirella and
Blain, Fred and
Pourmostafa Roshan Sharami, Javad and
Lepp, Lisa and
Manna, Chiara and
Rescigno, Argentina Anna and
Karakanta, Alina and
Rigouts Terryn, Ayla and
Lardelli, Manuel and
Resende, Natalia and
Murgolo, Elena and
Hackenbuchner, Jani{\c{c}}a and
Zaretskaya, Anna and
Espl{\`a}-Gomis, Miquel and
Etchegoyhen, Thierry and
Gromann, Dagmar and
Bawden, Rachel and
Haddow, Barry and
Szoc, Sara and
Forcada, Mikel and
Moniz, Helena",
booktitle = "Proceedings of the 26th Annual Conference of the {E}uropean Association for Machine Translation (Volume 1)",
month = jun,
year = "2026",
address = "Tilburg, The Netherlands",
publisher = "European Association for Machine Translation",
url = "https://aclanthology.org/2026.eamt-1.22/",
pages = "339--347",
ISBN = "9789403901411",
abstract = "Human evaluation of MT is essential but exhibits substantial annotator variability that limits evaluation reliability and super- vised learning. Rather than treating dis- agreement as noise or correcting it through protocol changes, we analyze its structure via learned severity classifiers. Across training regimes defined by base- line model reproducibility, we observe in- ternally coherent but mutually incompati- ble severity mappings: models trained on one regime produce confident predictions within that regime but reduced separability on the other. Margin{--}correctness analysis shows that instability is not uniformly low confidence; separability depends on align- ment between model-internalized and hu- man annotation regimes. These results indicate that unstable MT evaluation regions arise primarily from competing severity interpretations rather than intrinsic example difficulty. Model{--} annotator disagreement therefore provides a practical signal for identifying unstable evaluation regions during MT evaluation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="iakivchuk-2026-using">
<titleInfo>
<title>Using Model Disagreement to Identify Unstable Regions in MT Evaluation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Vitalii</namePart>
<namePart type="family">Iakivchuk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 1)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Dimitar</namePart>
<namePart type="family">Shterionov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eva</namePart>
<namePart type="family">Vanmassenhove</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mirella</namePart>
<namePart type="family">De Sisto</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fred</namePart>
<namePart type="family">Blain</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Javad</namePart>
<namePart type="family">Pourmostafa Roshan Sharami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lisa</namePart>
<namePart type="family">Lepp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chiara</namePart>
<namePart type="family">Manna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Argentina</namePart>
<namePart type="given">Anna</namePart>
<namePart type="family">Rescigno</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Karakanta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ayla</namePart>
<namePart type="family">Rigouts Terryn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Manuel</namePart>
<namePart type="family">Lardelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Natalia</namePart>
<namePart type="family">Resende</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Murgolo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Janiça</namePart>
<namePart type="family">Hackenbuchner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anna</namePart>
<namePart type="family">Zaretskaya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Miquel</namePart>
<namePart type="family">Esplà-Gomis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Thierry</namePart>
<namePart type="family">Etchegoyhen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dagmar</namePart>
<namePart type="family">Gromann</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rachel</namePart>
<namePart type="family">Bawden</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Barry</namePart>
<namePart type="family">Haddow</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sara</namePart>
<namePart type="family">Szoc</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mikel</namePart>
<namePart type="family">Forcada</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Helena</namePart>
<namePart type="family">Moniz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Association for Machine Translation</publisher>
<place>
<placeTerm type="text">Tilburg, The Netherlands</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">9789403901411</identifier>
</relatedItem>
<abstract>Human evaluation of MT is essential but exhibits substantial annotator variability that limits evaluation reliability and super- vised learning. Rather than treating dis- agreement as noise or correcting it through protocol changes, we analyze its structure via learned severity classifiers. Across training regimes defined by base- line model reproducibility, we observe in- ternally coherent but mutually incompati- ble severity mappings: models trained on one regime produce confident predictions within that regime but reduced separability on the other. Margin–correctness analysis shows that instability is not uniformly low confidence; separability depends on align- ment between model-internalized and hu- man annotation regimes. These results indicate that unstable MT evaluation regions arise primarily from competing severity interpretations rather than intrinsic example difficulty. Model– annotator disagreement therefore provides a practical signal for identifying unstable evaluation regions during MT evaluation.</abstract>
<identifier type="citekey">iakivchuk-2026-using</identifier>
<location>
<url>https://aclanthology.org/2026.eamt-1.22/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>339</start>
<end>347</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Using Model Disagreement to Identify Unstable Regions in MT Evaluation
%A Iakivchuk, Vitalii
%Y Shterionov, Dimitar
%Y Vanmassenhove, Eva
%Y De Sisto, Mirella
%Y Blain, Fred
%Y Pourmostafa Roshan Sharami, Javad
%Y Lepp, Lisa
%Y Manna, Chiara
%Y Rescigno, Argentina Anna
%Y Karakanta, Alina
%Y Rigouts Terryn, Ayla
%Y Lardelli, Manuel
%Y Resende, Natalia
%Y Murgolo, Elena
%Y Hackenbuchner, Janiça
%Y Zaretskaya, Anna
%Y Esplà-Gomis, Miquel
%Y Etchegoyhen, Thierry
%Y Gromann, Dagmar
%Y Bawden, Rachel
%Y Haddow, Barry
%Y Szoc, Sara
%Y Forcada, Mikel
%Y Moniz, Helena
%S Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 1)
%D 2026
%8 June
%I European Association for Machine Translation
%C Tilburg, The Netherlands
%@ 9789403901411
%F iakivchuk-2026-using
%X Human evaluation of MT is essential but exhibits substantial annotator variability that limits evaluation reliability and super- vised learning. Rather than treating dis- agreement as noise or correcting it through protocol changes, we analyze its structure via learned severity classifiers. Across training regimes defined by base- line model reproducibility, we observe in- ternally coherent but mutually incompati- ble severity mappings: models trained on one regime produce confident predictions within that regime but reduced separability on the other. Margin–correctness analysis shows that instability is not uniformly low confidence; separability depends on align- ment between model-internalized and hu- man annotation regimes. These results indicate that unstable MT evaluation regions arise primarily from competing severity interpretations rather than intrinsic example difficulty. Model– annotator disagreement therefore provides a practical signal for identifying unstable evaluation regions during MT evaluation.
%U https://aclanthology.org/2026.eamt-1.22/
%P 339-347
Markdown (Informal)
[Using Model Disagreement to Identify Unstable Regions in MT Evaluation](https://aclanthology.org/2026.eamt-1.22/) (Iakivchuk, EAMT 2026)
ACL