@inproceedings{dahan-etal-2026-metadoceval,
title = "{M}eta{D}oc{E}val: A Contrastive Framework for Evaluating Machine Translation Metrics at the Document-Level",
author = "Dahan, Nicolas and
Bawden, Rachel and
Yvon, Fran{\c{c}}ois",
editor = "Shterionov, Dimitar and
Vanmassenhove, Eva and
De Sisto, Mirella and
Blain, Fred and
Pourmostafa Roshan Sharami, Javad and
Lepp, Lisa and
Manna, Chiara and
Rescigno, Argentina Anna and
Karakanta, Alina and
Rigouts Terryn, Ayla and
Lardelli, Manuel and
Resende, Natalia and
Murgolo, Elena and
Hackenbuchner, Jani{\c{c}}a and
Zaretskaya, Anna and
Espl{\`a}-Gomis, Miquel and
Etchegoyhen, Thierry and
Gromann, Dagmar and
Bawden, Rachel and
Haddow, Barry and
Szoc, Sara and
Forcada, Mikel and
Moniz, Helena",
booktitle = "Proceedings of the 26th Annual Conference of the {E}uropean Association for Machine Translation (Volume 1)",
month = jun,
year = "2026",
address = "Tilburg, The Netherlands",
publisher = "European Association for Machine Translation",
url = "https://aclanthology.org/2026.eamt-1.19/",
pages = "268--303",
ISBN = "9789403901411",
abstract = "Recent advances in neural machine translation (MT) have spurred increased interest in evaluating translations beyond the sentence level, making it possible to assess discourse-level phenomena related to coherence and consistency. While existing metrics can be applied to multi-sentence spans, it remains unclear whether their scores truly capture document-level quality. We introduce MetaDocEval, an automatic contrastive test set for evaluating MT metrics across three language pairs (en{--}fr, en{--}es, en{--}de) when applied at the document-level. It targets a range of discourse-level phenomena and potential problems linked to translation at the document level. To evaluate how metrics behave as a function of context size, we apply them under a sliding-window protocol, varying the input from single sentences up to full documents. Our experiments show that no current metric genuinely captures document-level coherence: reference-based metrics overfit lexical overlap, reference+source metrics gain little from added context, reference-free encoders show brief context sensitivity before degrading on longer spans, and LLM-based scorers collapse beyond short inputs. A key finding is that reference access can be actively harmful for detecting discourse-level errors. Using short windows ({\ensuremath{\approx}} 3 sentences) offers the best trade-off between discourse error detection and score dilution."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="dahan-etal-2026-metadoceval">
<titleInfo>
<title>MetaDocEval: A Contrastive Framework for Evaluating Machine Translation Metrics at the Document-Level</title>
</titleInfo>
<name type="personal">
<namePart type="given">Nicolas</namePart>
<namePart type="family">Dahan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rachel</namePart>
<namePart type="family">Bawden</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">François</namePart>
<namePart type="family">Yvon</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 1)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Dimitar</namePart>
<namePart type="family">Shterionov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eva</namePart>
<namePart type="family">Vanmassenhove</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mirella</namePart>
<namePart type="family">De Sisto</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fred</namePart>
<namePart type="family">Blain</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Javad</namePart>
<namePart type="family">Pourmostafa Roshan Sharami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lisa</namePart>
<namePart type="family">Lepp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chiara</namePart>
<namePart type="family">Manna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Argentina</namePart>
<namePart type="given">Anna</namePart>
<namePart type="family">Rescigno</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Karakanta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ayla</namePart>
<namePart type="family">Rigouts Terryn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Manuel</namePart>
<namePart type="family">Lardelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Natalia</namePart>
<namePart type="family">Resende</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Murgolo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Janiça</namePart>
<namePart type="family">Hackenbuchner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anna</namePart>
<namePart type="family">Zaretskaya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Miquel</namePart>
<namePart type="family">Esplà-Gomis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Thierry</namePart>
<namePart type="family">Etchegoyhen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dagmar</namePart>
<namePart type="family">Gromann</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rachel</namePart>
<namePart type="family">Bawden</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Barry</namePart>
<namePart type="family">Haddow</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sara</namePart>
<namePart type="family">Szoc</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mikel</namePart>
<namePart type="family">Forcada</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Helena</namePart>
<namePart type="family">Moniz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Association for Machine Translation</publisher>
<place>
<placeTerm type="text">Tilburg, The Netherlands</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">9789403901411</identifier>
</relatedItem>
<abstract>Recent advances in neural machine translation (MT) have spurred increased interest in evaluating translations beyond the sentence level, making it possible to assess discourse-level phenomena related to coherence and consistency. While existing metrics can be applied to multi-sentence spans, it remains unclear whether their scores truly capture document-level quality. We introduce MetaDocEval, an automatic contrastive test set for evaluating MT metrics across three language pairs (en–fr, en–es, en–de) when applied at the document-level. It targets a range of discourse-level phenomena and potential problems linked to translation at the document level. To evaluate how metrics behave as a function of context size, we apply them under a sliding-window protocol, varying the input from single sentences up to full documents. Our experiments show that no current metric genuinely captures document-level coherence: reference-based metrics overfit lexical overlap, reference+source metrics gain little from added context, reference-free encoders show brief context sensitivity before degrading on longer spans, and LLM-based scorers collapse beyond short inputs. A key finding is that reference access can be actively harmful for detecting discourse-level errors. Using short windows (\ensuremath\approx 3 sentences) offers the best trade-off between discourse error detection and score dilution.</abstract>
<identifier type="citekey">dahan-etal-2026-metadoceval</identifier>
<location>
<url>https://aclanthology.org/2026.eamt-1.19/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>268</start>
<end>303</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T MetaDocEval: A Contrastive Framework for Evaluating Machine Translation Metrics at the Document-Level
%A Dahan, Nicolas
%A Bawden, Rachel
%A Yvon, François
%Y Shterionov, Dimitar
%Y Vanmassenhove, Eva
%Y De Sisto, Mirella
%Y Blain, Fred
%Y Pourmostafa Roshan Sharami, Javad
%Y Lepp, Lisa
%Y Manna, Chiara
%Y Rescigno, Argentina Anna
%Y Karakanta, Alina
%Y Rigouts Terryn, Ayla
%Y Lardelli, Manuel
%Y Resende, Natalia
%Y Murgolo, Elena
%Y Hackenbuchner, Janiça
%Y Zaretskaya, Anna
%Y Esplà-Gomis, Miquel
%Y Etchegoyhen, Thierry
%Y Gromann, Dagmar
%Y Bawden, Rachel
%Y Haddow, Barry
%Y Szoc, Sara
%Y Forcada, Mikel
%Y Moniz, Helena
%S Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 1)
%D 2026
%8 June
%I European Association for Machine Translation
%C Tilburg, The Netherlands
%@ 9789403901411
%F dahan-etal-2026-metadoceval
%X Recent advances in neural machine translation (MT) have spurred increased interest in evaluating translations beyond the sentence level, making it possible to assess discourse-level phenomena related to coherence and consistency. While existing metrics can be applied to multi-sentence spans, it remains unclear whether their scores truly capture document-level quality. We introduce MetaDocEval, an automatic contrastive test set for evaluating MT metrics across three language pairs (en–fr, en–es, en–de) when applied at the document-level. It targets a range of discourse-level phenomena and potential problems linked to translation at the document level. To evaluate how metrics behave as a function of context size, we apply them under a sliding-window protocol, varying the input from single sentences up to full documents. Our experiments show that no current metric genuinely captures document-level coherence: reference-based metrics overfit lexical overlap, reference+source metrics gain little from added context, reference-free encoders show brief context sensitivity before degrading on longer spans, and LLM-based scorers collapse beyond short inputs. A key finding is that reference access can be actively harmful for detecting discourse-level errors. Using short windows (\ensuremath\approx 3 sentences) offers the best trade-off between discourse error detection and score dilution.
%U https://aclanthology.org/2026.eamt-1.19/
%P 268-303
Markdown (Informal)
[MetaDocEval: A Contrastive Framework for Evaluating Machine Translation Metrics at the Document-Level](https://aclanthology.org/2026.eamt-1.19/) (Dahan et al., EAMT 2026)
ACL