@inproceedings{gonzalez-saez-etal-2026-come,
title = "{COME}-{ALP}s: Coreference Annotation with {ME}rging Heuristics Using {AL}ignment-based Projection in Parallel Corpora",
author = "Gonzalez Saez, Gabriela Nicole and
Nakhle, Mariam and
Kholosha, Illia and
Atherly, Rachel and
Dinarelli, Marco",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.133/",
doi = "10.63317/2ohkaq9ps5hd",
pages = "1688--1695",
abstract = "Multi-lingual, parallel datasets annotated with discourse phenomena like coreferences are a rare resource. These datasets are useful and informative to evaluate models for NLP tasks taking long contextual information into account, as proved by the large literature published in the last couple of years on e.g. Context-Aware Neural Machine Translation (CA-NMT). Inspired by resources published in previous work, in this paper we propose an automated procedure to annotate multi-lingual, parallel data with coreferences. Through the use of accurate alignment and coreference annotation tools, we project the annotation from English data, where tools are most often more accurate, to one or more target languages. We apply some consistency constraints to obtain more accurate annotations on both source and target side. Using our procedure we generated two new resources that can be used for evaluating CA-NMT models. One starting from the well-known TED Talk{'}s data released for the IWSLT17 shared task, where we project the annotation from English to target languages as diverse as French, German and Chinese. The second resource is derived from the WMT24 shared task, consisting of news domain data in the same set of target languages. We release these resources, as well as the code framework for applying our annotation procedure, to the community."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="gonzalez-saez-etal-2026-come">
<titleInfo>
<title>COME-ALPs: Coreference Annotation with MErging Heuristics Using ALignment-based Projection in Parallel Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Gabriela</namePart>
<namePart type="given">Nicole</namePart>
<namePart type="family">Gonzalez Saez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mariam</namePart>
<namePart type="family">Nakhle</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Illia</namePart>
<namePart type="family">Kholosha</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rachel</namePart>
<namePart type="family">Atherly</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marco</namePart>
<namePart type="family">Dinarelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Multi-lingual, parallel datasets annotated with discourse phenomena like coreferences are a rare resource. These datasets are useful and informative to evaluate models for NLP tasks taking long contextual information into account, as proved by the large literature published in the last couple of years on e.g. Context-Aware Neural Machine Translation (CA-NMT). Inspired by resources published in previous work, in this paper we propose an automated procedure to annotate multi-lingual, parallel data with coreferences. Through the use of accurate alignment and coreference annotation tools, we project the annotation from English data, where tools are most often more accurate, to one or more target languages. We apply some consistency constraints to obtain more accurate annotations on both source and target side. Using our procedure we generated two new resources that can be used for evaluating CA-NMT models. One starting from the well-known TED Talk’s data released for the IWSLT17 shared task, where we project the annotation from English to target languages as diverse as French, German and Chinese. The second resource is derived from the WMT24 shared task, consisting of news domain data in the same set of target languages. We release these resources, as well as the code framework for applying our annotation procedure, to the community.</abstract>
<identifier type="citekey">gonzalez-saez-etal-2026-come</identifier>
<identifier type="doi">10.63317/2ohkaq9ps5hd</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.133/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>1688</start>
<end>1695</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T COME-ALPs: Coreference Annotation with MErging Heuristics Using ALignment-based Projection in Parallel Corpora
%A Gonzalez Saez, Gabriela Nicole
%A Nakhle, Mariam
%A Kholosha, Illia
%A Atherly, Rachel
%A Dinarelli, Marco
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F gonzalez-saez-etal-2026-come
%X Multi-lingual, parallel datasets annotated with discourse phenomena like coreferences are a rare resource. These datasets are useful and informative to evaluate models for NLP tasks taking long contextual information into account, as proved by the large literature published in the last couple of years on e.g. Context-Aware Neural Machine Translation (CA-NMT). Inspired by resources published in previous work, in this paper we propose an automated procedure to annotate multi-lingual, parallel data with coreferences. Through the use of accurate alignment and coreference annotation tools, we project the annotation from English data, where tools are most often more accurate, to one or more target languages. We apply some consistency constraints to obtain more accurate annotations on both source and target side. Using our procedure we generated two new resources that can be used for evaluating CA-NMT models. One starting from the well-known TED Talk’s data released for the IWSLT17 shared task, where we project the annotation from English to target languages as diverse as French, German and Chinese. The second resource is derived from the WMT24 shared task, consisting of news domain data in the same set of target languages. We release these resources, as well as the code framework for applying our annotation procedure, to the community.
%R 10.63317/2ohkaq9ps5hd
%U https://aclanthology.org/2026.lrec-1.133/
%U https://doi.org/10.63317/2ohkaq9ps5hd
%P 1688-1695
Markdown (Informal)
[COME-ALPs: Coreference Annotation with MErging Heuristics Using ALignment-based Projection in Parallel Corpora](https://aclanthology.org/2026.lrec-1.133/) (Gonzalez Saez et al., LREC 2026)
ACL