@article{tiedemann-luo-2026-opensubtitles2024,
title = "{O}pen{S}ubtitles2024: A Massively Parallel Dataset of Movie Subtitles for {MT} Development and Evaluation",
author = "Tiedemann, Joerg and
Luo, Hengyu",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
journal = "International Conference on Language Resources and Evaluation",
volume = "main",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/ingest-lrec/2026.lrec-main.700/",
pages = "8897--8907",
abstract = "This paper introduces OpenSubtitles2024, a massively parallel dataset compiled from translated subtitles. The collection includes an extensive collection of aligned training data based on user-contributed subtitles derived from OpenSubtitles.org and a dedicated held-out dataset for development and evaluation of machine translation and multilingual language models. The collection provides an increased language coverage and doubles the size of the previous edition. Furthermore, a careful procedure was applied to reserve a subset of the most recent subtitles for system development and evaluation. The collection covers 92 languages and language variants, aligned in over 3,000 bitexts containing 40 billion tokens in 7.7 million subtitle files. The test set comprises 2,022 language pairs. In addition, we also provide a multi-parallel test set that refers to a subset of the held-out data with synchronized alignments across 40 languages and 15 subtitles."
}Markdown (Informal)
[OpenSubtitles2024: A Massively Parallel Dataset of Movie Subtitles for MT Development and Evaluation](https://preview.aclanthology.org/ingest-lrec/2026.lrec-main.700/) (Tiedemann & Luo, LREC 2026)
ACL