@inproceedings{hingrajiya-etal-2026-indicdisco,
title = "{I}ndic{DISCO}-{MT}: A Discourse-Centric Benchmark for Evaluating Discourse Phenomena in {I}ndian Language Machine Translation",
author = "Hingrajiya, Heli and
Bairi, Vennela and
Mujadia, Vandan and
Sharma, Dipti and
Krishnamurthy, Parameswari and
Varma, Vasudeva",
editor = "Shterionov, Dimitar and
Vanmassenhove, Eva and
De Sisto, Mirella and
Blain, Fred and
Pourmostafa Roshan Sharami, Javad and
Lepp, Lisa and
Manna, Chiara and
Rescigno, Argentina Anna and
Karakanta, Alina and
Rigouts Terryn, Ayla and
Lardelli, Manuel and
Resende, Natalia and
Murgolo, Elena and
Hackenbuchner, Jani{\c{c}}a and
Zaretskaya, Anna and
Espl{\`a}-Gomis, Miquel and
Etchegoyhen, Thierry and
Gromann, Dagmar and
Bawden, Rachel and
Haddow, Barry and
Szoc, Sara and
Forcada, Mikel and
Moniz, Helena",
booktitle = "Proceedings of the 26th Annual Conference of the {E}uropean Association for Machine Translation (Volume 1)",
month = jun,
year = "2026",
address = "Tilburg, The Netherlands",
publisher = "European Association for Machine Translation",
url = "https://aclanthology.org/2026.eamt-1.15/",
pages = "190--204",
ISBN = "9789403901411",
abstract = "Discourse-level translation remains a major challenge for machine translation (MT) systems, particularly for translation from Indian languages to English. This difficulty arises due to factors such as rich morphology, diverse syntactic structures, unmarked gender distinctions in pronouns, and the limited availability of discourse-aware training data. Existing evaluation benchmarks primarily focus on sentence-level translation quality and fail to capture important discourse phenomena such as pronoun resolution and lexical cohesion. To address this gap, we introduce IndicDISCO-MT, a parallel benchmark dataset covering translations from eight Indian languages such as Bengali, Gujarati, Hindi, Marathi, Kannada, Tamil, Telugu and Urdu to English. On top of this dataset, we are first to introduce DiscoAlign, a human annotated word to word alignment benchmark dataset, that captures correspondences between source and target words across languages. In addition, we propose two evaluation benchmarks, ProAlign and LexiAlign, designed to specifically assess the ability of large language models (LLMs) and MT systems to handle personal pronouns and lexical cohesion. Our evaluation of recent LLMs and MT systems on these benchmarks shows that although models achieve high overall translation quality, they still struggle to accurately preserve discourse-level phenomena. The proposed benchmarks provide a systematic framework for evaluating discourse aware translation and can facilitate the development of MT systems that generate more coherent and contextually consistent translations."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hingrajiya-etal-2026-indicdisco">
<titleInfo>
<title>IndicDISCO-MT: A Discourse-Centric Benchmark for Evaluating Discourse Phenomena in Indian Language Machine Translation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Heli</namePart>
<namePart type="family">Hingrajiya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vennela</namePart>
<namePart type="family">Bairi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vandan</namePart>
<namePart type="family">Mujadia</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dipti</namePart>
<namePart type="family">Sharma</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Parameswari</namePart>
<namePart type="family">Krishnamurthy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vasudeva</namePart>
<namePart type="family">Varma</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 1)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Dimitar</namePart>
<namePart type="family">Shterionov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eva</namePart>
<namePart type="family">Vanmassenhove</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mirella</namePart>
<namePart type="family">De Sisto</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fred</namePart>
<namePart type="family">Blain</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Javad</namePart>
<namePart type="family">Pourmostafa Roshan Sharami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lisa</namePart>
<namePart type="family">Lepp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chiara</namePart>
<namePart type="family">Manna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Argentina</namePart>
<namePart type="given">Anna</namePart>
<namePart type="family">Rescigno</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Karakanta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ayla</namePart>
<namePart type="family">Rigouts Terryn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Manuel</namePart>
<namePart type="family">Lardelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Natalia</namePart>
<namePart type="family">Resende</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Murgolo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Janiça</namePart>
<namePart type="family">Hackenbuchner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anna</namePart>
<namePart type="family">Zaretskaya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Miquel</namePart>
<namePart type="family">Esplà-Gomis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Thierry</namePart>
<namePart type="family">Etchegoyhen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dagmar</namePart>
<namePart type="family">Gromann</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rachel</namePart>
<namePart type="family">Bawden</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Barry</namePart>
<namePart type="family">Haddow</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sara</namePart>
<namePart type="family">Szoc</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mikel</namePart>
<namePart type="family">Forcada</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Helena</namePart>
<namePart type="family">Moniz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Association for Machine Translation</publisher>
<place>
<placeTerm type="text">Tilburg, The Netherlands</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">9789403901411</identifier>
</relatedItem>
<abstract>Discourse-level translation remains a major challenge for machine translation (MT) systems, particularly for translation from Indian languages to English. This difficulty arises due to factors such as rich morphology, diverse syntactic structures, unmarked gender distinctions in pronouns, and the limited availability of discourse-aware training data. Existing evaluation benchmarks primarily focus on sentence-level translation quality and fail to capture important discourse phenomena such as pronoun resolution and lexical cohesion. To address this gap, we introduce IndicDISCO-MT, a parallel benchmark dataset covering translations from eight Indian languages such as Bengali, Gujarati, Hindi, Marathi, Kannada, Tamil, Telugu and Urdu to English. On top of this dataset, we are first to introduce DiscoAlign, a human annotated word to word alignment benchmark dataset, that captures correspondences between source and target words across languages. In addition, we propose two evaluation benchmarks, ProAlign and LexiAlign, designed to specifically assess the ability of large language models (LLMs) and MT systems to handle personal pronouns and lexical cohesion. Our evaluation of recent LLMs and MT systems on these benchmarks shows that although models achieve high overall translation quality, they still struggle to accurately preserve discourse-level phenomena. The proposed benchmarks provide a systematic framework for evaluating discourse aware translation and can facilitate the development of MT systems that generate more coherent and contextually consistent translations.</abstract>
<identifier type="citekey">hingrajiya-etal-2026-indicdisco</identifier>
<location>
<url>https://aclanthology.org/2026.eamt-1.15/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>190</start>
<end>204</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T IndicDISCO-MT: A Discourse-Centric Benchmark for Evaluating Discourse Phenomena in Indian Language Machine Translation
%A Hingrajiya, Heli
%A Bairi, Vennela
%A Mujadia, Vandan
%A Sharma, Dipti
%A Krishnamurthy, Parameswari
%A Varma, Vasudeva
%Y Shterionov, Dimitar
%Y Vanmassenhove, Eva
%Y De Sisto, Mirella
%Y Blain, Fred
%Y Pourmostafa Roshan Sharami, Javad
%Y Lepp, Lisa
%Y Manna, Chiara
%Y Rescigno, Argentina Anna
%Y Karakanta, Alina
%Y Rigouts Terryn, Ayla
%Y Lardelli, Manuel
%Y Resende, Natalia
%Y Murgolo, Elena
%Y Hackenbuchner, Janiça
%Y Zaretskaya, Anna
%Y Esplà-Gomis, Miquel
%Y Etchegoyhen, Thierry
%Y Gromann, Dagmar
%Y Bawden, Rachel
%Y Haddow, Barry
%Y Szoc, Sara
%Y Forcada, Mikel
%Y Moniz, Helena
%S Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 1)
%D 2026
%8 June
%I European Association for Machine Translation
%C Tilburg, The Netherlands
%@ 9789403901411
%F hingrajiya-etal-2026-indicdisco
%X Discourse-level translation remains a major challenge for machine translation (MT) systems, particularly for translation from Indian languages to English. This difficulty arises due to factors such as rich morphology, diverse syntactic structures, unmarked gender distinctions in pronouns, and the limited availability of discourse-aware training data. Existing evaluation benchmarks primarily focus on sentence-level translation quality and fail to capture important discourse phenomena such as pronoun resolution and lexical cohesion. To address this gap, we introduce IndicDISCO-MT, a parallel benchmark dataset covering translations from eight Indian languages such as Bengali, Gujarati, Hindi, Marathi, Kannada, Tamil, Telugu and Urdu to English. On top of this dataset, we are first to introduce DiscoAlign, a human annotated word to word alignment benchmark dataset, that captures correspondences between source and target words across languages. In addition, we propose two evaluation benchmarks, ProAlign and LexiAlign, designed to specifically assess the ability of large language models (LLMs) and MT systems to handle personal pronouns and lexical cohesion. Our evaluation of recent LLMs and MT systems on these benchmarks shows that although models achieve high overall translation quality, they still struggle to accurately preserve discourse-level phenomena. The proposed benchmarks provide a systematic framework for evaluating discourse aware translation and can facilitate the development of MT systems that generate more coherent and contextually consistent translations.
%U https://aclanthology.org/2026.eamt-1.15/
%P 190-204
Markdown (Informal)
[IndicDISCO-MT: A Discourse-Centric Benchmark for Evaluating Discourse Phenomena in Indian Language Machine Translation](https://aclanthology.org/2026.eamt-1.15/) (Hingrajiya et al., EAMT 2026)
ACL