@inproceedings{lacunza-etal-2026-acadata,
title = "{ACAD}ata: Parallel Dataset of Academic Data for Machine Translation",
author = "Lacunza, I{\~n}aki and
Garcia Gilabert, Javier and
De Luca Fornaciari, Francesca and
Aula-Blasco, Javier and
Gonzalez-Agirre, Aitor and
Melero, Maite and
Villegas, Marta",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.671/",
doi = "10.63317/4fkj9gvuqsdd",
pages = "8498--8519",
abstract = "We present ACAData, a high-quality parallel dataset for academic translation, that consists of two subsets: ACAD-Train, which contains approximately 1.5 million human-generated paragraph pairs across 12 languages, and ACAD-Bench, a curated evaluation set of almost 6,000 translations covering 12 directions. To validate its usefulness, we fine-tune two Large Language Models (LLMs) on ACAD-Train and benchmark them on ACAD-Bench against specialized machine-translation systems, general-purpose, open-weight LLMs, and several large-scale proprietary models. Experimental results demonstrate that fine tuning on ACAD-Train leads to improvements in academic translation quality by +6.1 and +12.4 d-BLEU points on average for 7B and 2B models respectively, while also improving long-context translation in a general domain by up to 24.9{\%} when translating out of English. The fine-tuned top-performing model surpasses the best proprietary and open-weight models on the academic translation domain. By releasing ACAD-Train, ACAD-Bench and the fine-tuned models, we provide the community with a valuable resource to advance research in the academic domain and long-context translation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="lacunza-etal-2026-acadata">
<titleInfo>
<title>ACAData: Parallel Dataset of Academic Data for Machine Translation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Iñaki</namePart>
<namePart type="family">Lacunza</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Javier</namePart>
<namePart type="family">Garcia Gilabert</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Francesca</namePart>
<namePart type="family">De Luca Fornaciari</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Javier</namePart>
<namePart type="family">Aula-Blasco</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aitor</namePart>
<namePart type="family">Gonzalez-Agirre</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maite</namePart>
<namePart type="family">Melero</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marta</namePart>
<namePart type="family">Villegas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present ACAData, a high-quality parallel dataset for academic translation, that consists of two subsets: ACAD-Train, which contains approximately 1.5 million human-generated paragraph pairs across 12 languages, and ACAD-Bench, a curated evaluation set of almost 6,000 translations covering 12 directions. To validate its usefulness, we fine-tune two Large Language Models (LLMs) on ACAD-Train and benchmark them on ACAD-Bench against specialized machine-translation systems, general-purpose, open-weight LLMs, and several large-scale proprietary models. Experimental results demonstrate that fine tuning on ACAD-Train leads to improvements in academic translation quality by +6.1 and +12.4 d-BLEU points on average for 7B and 2B models respectively, while also improving long-context translation in a general domain by up to 24.9% when translating out of English. The fine-tuned top-performing model surpasses the best proprietary and open-weight models on the academic translation domain. By releasing ACAD-Train, ACAD-Bench and the fine-tuned models, we provide the community with a valuable resource to advance research in the academic domain and long-context translation.</abstract>
<identifier type="citekey">lacunza-etal-2026-acadata</identifier>
<identifier type="doi">10.63317/4fkj9gvuqsdd</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.671/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>8498</start>
<end>8519</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T ACAData: Parallel Dataset of Academic Data for Machine Translation
%A Lacunza, Iñaki
%A Garcia Gilabert, Javier
%A De Luca Fornaciari, Francesca
%A Aula-Blasco, Javier
%A Gonzalez-Agirre, Aitor
%A Melero, Maite
%A Villegas, Marta
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F lacunza-etal-2026-acadata
%X We present ACAData, a high-quality parallel dataset for academic translation, that consists of two subsets: ACAD-Train, which contains approximately 1.5 million human-generated paragraph pairs across 12 languages, and ACAD-Bench, a curated evaluation set of almost 6,000 translations covering 12 directions. To validate its usefulness, we fine-tune two Large Language Models (LLMs) on ACAD-Train and benchmark them on ACAD-Bench against specialized machine-translation systems, general-purpose, open-weight LLMs, and several large-scale proprietary models. Experimental results demonstrate that fine tuning on ACAD-Train leads to improvements in academic translation quality by +6.1 and +12.4 d-BLEU points on average for 7B and 2B models respectively, while also improving long-context translation in a general domain by up to 24.9% when translating out of English. The fine-tuned top-performing model surpasses the best proprietary and open-weight models on the academic translation domain. By releasing ACAD-Train, ACAD-Bench and the fine-tuned models, we provide the community with a valuable resource to advance research in the academic domain and long-context translation.
%R 10.63317/4fkj9gvuqsdd
%U https://aclanthology.org/2026.lrec-1.671/
%U https://doi.org/10.63317/4fkj9gvuqsdd
%P 8498-8519
Markdown (Informal)
[ACAData: Parallel Dataset of Academic Data for Machine Translation](https://aclanthology.org/2026.lrec-1.671/) (Lacunza et al., LREC 2026)
ACL
- Iñaki Lacunza, Javier Garcia Gilabert, Francesca De Luca Fornaciari, Javier Aula-Blasco, Aitor Gonzalez-Agirre, Maite Melero, and Marta Villegas. 2026. ACAData: Parallel Dataset of Academic Data for Machine Translation. In Proceedings of the Fifteenth Language Resources and Evaluation Conference, pages 8498–8519, Palma de Mallorca, Spain. ELRA Language Resource Association.