@inproceedings{doi-etal-2020-tddc,
title = "{TDDC}: Timely Disclosure Documents Corpus",
author = "Doi, Nobushige and
Oda, Yusuke and
Nakazawa, Toshiaki",
editor = "Calzolari, Nicoletta and
B{\'e}chet, Fr{\'e}d{\'e}ric and
Blache, Philippe and
Choukri, Khalid and
Cieri, Christopher and
Declerck, Thierry and
Goggi, Sara and
Isahara, Hitoshi and
Maegaard, Bente and
Mariani, Joseph and
Mazo, H{\'e}l{\`e}ne and
Moreno, Asuncion and
Odijk, Jan and
Piperidis, Stelios",
booktitle = "Proceedings of the Twelfth Language Resources and Evaluation Conference",
month = may,
year = "2020",
address = "Marseille, France",
publisher = "European Language Resources Association",
url = "https://preview.aclanthology.org/fix-sig-urls/2020.lrec-1.459/",
pages = "3719--3726",
language = "eng",
ISBN = "979-10-95546-34-4",
abstract = "In this paper, we describe the details of the Timely Disclosure Documents Corpus (TDDC). TDDC was prepared by manually aligning the sentences from past Japanese and English timely disclosure documents in PDF format published by companies listed on the Tokyo Stock Exchange. TDDC consists of approximately 1.4 million parallel sentences in Japanese and English. TDDC was used as the official dataset for the 6th Workshop on Asian Translation to encourage the development of machine translation."
}
Markdown (Informal)
[TDDC: Timely Disclosure Documents Corpus](https://preview.aclanthology.org/fix-sig-urls/2020.lrec-1.459/) (Doi et al., LREC 2020)
ACL
- Nobushige Doi, Yusuke Oda, and Toshiaki Nakazawa. 2020. TDDC: Timely Disclosure Documents Corpus. In Proceedings of the Twelfth Language Resources and Evaluation Conference, pages 3719–3726, Marseille, France. European Language Resources Association.