@inproceedings{peng-etal-2026-parallel,
title = "Parallel Corpora of Scholarly Documents for {E}nglish-{F}rench Machine Translation",
author = {Peng, Ziqian and
Zhu, Lichao and
Bawden, Rachel and
B{\'e}nard, Maud and
de la Clergerie, {\'E}ric and
Huguin, Mathilde and
K{\"u}bler, Natalie and
Lerner, Paul and
Mestivier, Alexandra and
Yvon, Fran{\c{c}}ois},
editor = "Rapp, Reinhard and
Terryn, Ayla Rigouts and
Sharoff, Serge and
Zweigenbaum, Pierre",
booktitle = "Proceedings of the 19th Workshop on Building and Using Comparable Corpora ({BUCC})",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.bucc-1.10/",
doi = "10.63317/2jm9pkbjkg95",
pages = "84--95",
abstract = "The growing ability of large language models (LLMs) to process long-range context opens new perspectives for document-level machine translation (MT), especially in scholarly communication. In fact, translating scholarly texts requires to integrate both local and long-range contextual information to ensure the consistency and coherence across the full document. However, document-level parallel corpora for such text types remain scarce, limiting both evaluation and domain adaptation of MT systems for this task. To address this gap, we introduce ParaEPS (Earth and Planetary Sciences Bilingual Corpus) and ParaNLP (Natural Language Processing Bilingual Corpus), two new parallel corpora covering 14k abstracts and 103 full-length articles in two scientific domains to be used for fine-tuning and evaluation purposes. We compare the performance of eight MT systems on these test sets and find that fine-tuning on document-level data closes the gap between open systems based on Large Language Models (LLMs) and commercial systems. We also find that the performance of recent LLMs can worsen when translating full articles instead of translating them on a per paragraph basisfine-tuning. These experiments underscore the need for corpora such as ParaEPS and ParaNLP."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="peng-etal-2026-parallel">
<titleInfo>
<title>Parallel Corpora of Scholarly Documents for English-French Machine Translation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ziqian</namePart>
<namePart type="family">Peng</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lichao</namePart>
<namePart type="family">Zhu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rachel</namePart>
<namePart type="family">Bawden</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maud</namePart>
<namePart type="family">Bénard</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Éric</namePart>
<namePart type="family">de la Clergerie</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mathilde</namePart>
<namePart type="family">Huguin</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Natalie</namePart>
<namePart type="family">Kübler</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paul</namePart>
<namePart type="family">Lerner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alexandra</namePart>
<namePart type="family">Mestivier</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">François</namePart>
<namePart type="family">Yvon</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 19th Workshop on Building and Using Comparable Corpora (BUCC)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Reinhard</namePart>
<namePart type="family">Rapp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ayla</namePart>
<namePart type="given">Rigouts</namePart>
<namePart type="family">Terryn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Serge</namePart>
<namePart type="family">Sharoff</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pierre</namePart>
<namePart type="family">Zweigenbaum</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The growing ability of large language models (LLMs) to process long-range context opens new perspectives for document-level machine translation (MT), especially in scholarly communication. In fact, translating scholarly texts requires to integrate both local and long-range contextual information to ensure the consistency and coherence across the full document. However, document-level parallel corpora for such text types remain scarce, limiting both evaluation and domain adaptation of MT systems for this task. To address this gap, we introduce ParaEPS (Earth and Planetary Sciences Bilingual Corpus) and ParaNLP (Natural Language Processing Bilingual Corpus), two new parallel corpora covering 14k abstracts and 103 full-length articles in two scientific domains to be used for fine-tuning and evaluation purposes. We compare the performance of eight MT systems on these test sets and find that fine-tuning on document-level data closes the gap between open systems based on Large Language Models (LLMs) and commercial systems. We also find that the performance of recent LLMs can worsen when translating full articles instead of translating them on a per paragraph basisfine-tuning. These experiments underscore the need for corpora such as ParaEPS and ParaNLP.</abstract>
<identifier type="citekey">peng-etal-2026-parallel</identifier>
<identifier type="doi">10.63317/2jm9pkbjkg95</identifier>
<location>
<url>https://aclanthology.org/2026.bucc-1.10/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>84</start>
<end>95</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Parallel Corpora of Scholarly Documents for English-French Machine Translation
%A Peng, Ziqian
%A Zhu, Lichao
%A Bawden, Rachel
%A Bénard, Maud
%A de la Clergerie, Éric
%A Huguin, Mathilde
%A Kübler, Natalie
%A Lerner, Paul
%A Mestivier, Alexandra
%A Yvon, François
%Y Rapp, Reinhard
%Y Terryn, Ayla Rigouts
%Y Sharoff, Serge
%Y Zweigenbaum, Pierre
%S Proceedings of the 19th Workshop on Building and Using Comparable Corpora (BUCC)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F peng-etal-2026-parallel
%X The growing ability of large language models (LLMs) to process long-range context opens new perspectives for document-level machine translation (MT), especially in scholarly communication. In fact, translating scholarly texts requires to integrate both local and long-range contextual information to ensure the consistency and coherence across the full document. However, document-level parallel corpora for such text types remain scarce, limiting both evaluation and domain adaptation of MT systems for this task. To address this gap, we introduce ParaEPS (Earth and Planetary Sciences Bilingual Corpus) and ParaNLP (Natural Language Processing Bilingual Corpus), two new parallel corpora covering 14k abstracts and 103 full-length articles in two scientific domains to be used for fine-tuning and evaluation purposes. We compare the performance of eight MT systems on these test sets and find that fine-tuning on document-level data closes the gap between open systems based on Large Language Models (LLMs) and commercial systems. We also find that the performance of recent LLMs can worsen when translating full articles instead of translating them on a per paragraph basisfine-tuning. These experiments underscore the need for corpora such as ParaEPS and ParaNLP.
%R 10.63317/2jm9pkbjkg95
%U https://aclanthology.org/2026.bucc-1.10/
%U https://doi.org/10.63317/2jm9pkbjkg95
%P 84-95
Markdown (Informal)
[Parallel Corpora of Scholarly Documents for English-French Machine Translation](https://aclanthology.org/2026.bucc-1.10/) (Peng et al., BUCC 2026)
ACL
- Ziqian Peng, Lichao Zhu, Rachel Bawden, Maud Bénard, Éric de la Clergerie, Mathilde Huguin, Natalie Kübler, Paul Lerner, Alexandra Mestivier, and François Yvon. 2026. Parallel Corpora of Scholarly Documents for English-French Machine Translation. In Proceedings of the 19th Workshop on Building and Using Comparable Corpora (BUCC), pages 84–95, Palma de Mallorca, Spain. ELRA Language Resources Association (ELRA).