@inproceedings{zhou-yoshinaga-2025-tasc,
title = "A-{TASC}: {A}sian {TED}-Based Automatic Subtitling Corpus",
author = "Zhou, Yuhan and
Yoshinaga, Naoki",
editor = "Che, Wanxiang and
Nabende, Joyce and
Shutova, Ekaterina and
Pilehvar, Mohammad Taher",
booktitle = "Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)",
month = jul,
year = "2025",
address = "Vienna, Austria",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/ingestion-acl-25/2025.acl-long.157/",
pages = "3135--3148",
ISBN = "979-8-89176-251-0",
abstract = "Subtitles play a crucial role in improving the accessibility of the vast amount of audiovisual content available on the Internet, allowing audiences worldwide to comprehend and engage with this content in various languages. Automatic subtitling (AS) systems are essential for alleviating the substantial workload of human transcribers and translators. However, existing AS corpora and the primary metric SubER focus on European languages. This paper introduces A-TASC, an Asian TED-based automatic subtitling corpus derived from English TED Talks, comprising nearly 800 hours of audio segments, aligned English transcripts, and subtitles in Chinese, Japanese, Korean, and Vietnamese. We then present SacreSubER, a modification of SubER, to enable the reliable evaluation of subtitle quality for languages without explicit word boundaries. Experimental results, using both end-to-end systems and pipeline approaches built on strong ASR and LLM components, validate the quality of the proposed corpus and reveal differences in AS performance between European and Asian languages. The code to build our corpus is released."
}
Markdown (Informal)
[A-TASC: Asian TED-Based Automatic Subtitling Corpus](https://preview.aclanthology.org/ingestion-acl-25/2025.acl-long.157/) (Zhou & Yoshinaga, ACL 2025)
ACL
- Yuhan Zhou and Naoki Yoshinaga. 2025. A-TASC: Asian TED-Based Automatic Subtitling Corpus. In Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pages 3135–3148, Vienna, Austria. Association for Computational Linguistics.