@inproceedings{tiedemann-luo-2026-opensubtitles2024,
title = "{O}pen{S}ubtitles2024: A Massively Parallel Dataset of Movie Subtitles for {MT} Development and Evaluation",
author = "Tiedemann, Joerg and
Luo, Hengyu",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.700/",
doi = "10.63317/4ivg578ub2ob",
pages = "8897--8907",
abstract = "This paper introduces OpenSubtitles2024, a massively parallel dataset compiled from translated subtitles. The collection includes an extensive collection of aligned training data based on user-contributed subtitles derived from OpenSubtitles.org and a dedicated held-out dataset for development and evaluation of machine translation and multilingual language models. The collection provides an increased language coverage and doubles the size of the previous edition. Furthermore, a careful procedure was applied to reserve a subset of the most recent subtitles for system development and evaluation. The collection covers 92 languages and language variants, aligned in over 3,000 bitexts containing 40 billion tokens in 7.7 million subtitle files. The test set comprises 2,022 language pairs. In addition, we also provide a multi-parallel test set that refers to a subset of the held-out data with synchronized alignments across 40 languages and 15 subtitles."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="tiedemann-luo-2026-opensubtitles2024">
<titleInfo>
<title>OpenSubtitles2024: A Massively Parallel Dataset of Movie Subtitles for MT Development and Evaluation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joerg</namePart>
<namePart type="family">Tiedemann</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hengyu</namePart>
<namePart type="family">Luo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper introduces OpenSubtitles2024, a massively parallel dataset compiled from translated subtitles. The collection includes an extensive collection of aligned training data based on user-contributed subtitles derived from OpenSubtitles.org and a dedicated held-out dataset for development and evaluation of machine translation and multilingual language models. The collection provides an increased language coverage and doubles the size of the previous edition. Furthermore, a careful procedure was applied to reserve a subset of the most recent subtitles for system development and evaluation. The collection covers 92 languages and language variants, aligned in over 3,000 bitexts containing 40 billion tokens in 7.7 million subtitle files. The test set comprises 2,022 language pairs. In addition, we also provide a multi-parallel test set that refers to a subset of the held-out data with synchronized alignments across 40 languages and 15 subtitles.</abstract>
<identifier type="citekey">tiedemann-luo-2026-opensubtitles2024</identifier>
<identifier type="doi">10.63317/4ivg578ub2ob</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.700/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>8897</start>
<end>8907</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T OpenSubtitles2024: A Massively Parallel Dataset of Movie Subtitles for MT Development and Evaluation
%A Tiedemann, Joerg
%A Luo, Hengyu
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F tiedemann-luo-2026-opensubtitles2024
%X This paper introduces OpenSubtitles2024, a massively parallel dataset compiled from translated subtitles. The collection includes an extensive collection of aligned training data based on user-contributed subtitles derived from OpenSubtitles.org and a dedicated held-out dataset for development and evaluation of machine translation and multilingual language models. The collection provides an increased language coverage and doubles the size of the previous edition. Furthermore, a careful procedure was applied to reserve a subset of the most recent subtitles for system development and evaluation. The collection covers 92 languages and language variants, aligned in over 3,000 bitexts containing 40 billion tokens in 7.7 million subtitle files. The test set comprises 2,022 language pairs. In addition, we also provide a multi-parallel test set that refers to a subset of the held-out data with synchronized alignments across 40 languages and 15 subtitles.
%R 10.63317/4ivg578ub2ob
%U https://aclanthology.org/2026.lrec-1.700/
%U https://doi.org/10.63317/4ivg578ub2ob
%P 8897-8907
Markdown (Informal)
[OpenSubtitles2024: A Massively Parallel Dataset of Movie Subtitles for MT Development and Evaluation](https://aclanthology.org/2026.lrec-1.700/) (Tiedemann & Luo, LREC 2026)
ACL