@inproceedings{gretter-2014-euronews,
title = "{E}uronews: a multilingual speech corpus for {ASR}",
author = "Gretter, Roberto",
editor = "Calzolari, Nicoletta and
Choukri, Khalid and
Declerck, Thierry and
Loftsson, Hrafn and
Maegaard, Bente and
Mariani, Joseph and
Moreno, Asuncion and
Odijk, Jan and
Piperidis, Stelios",
booktitle = "Proceedings of the Ninth International Conference on Language Resources and Evaluation ({LREC}'14)",
month = may,
year = "2014",
address = "Reykjavik, Iceland",
publisher = "European Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/fix-sig-urls/L14-1546/",
pages = "2635--2638",
abstract = "In this paper we present a multilingual speech corpus, designed for Automatic Speech Recognition (ASR) purposes. Data come from the portal Euronews and were acquired both from the Web and from TV. The corpus includes data in 10 languages (Arabic, English, French, German, Italian, Polish, Portuguese, Russian, Spanish and Turkish) and was designed both to train AMs and to evaluate ASR performance. For each language, the corpus is composed of about 100 hours of speech for training (60 for Polish) and about 4 hours, manually transcribed, for testing. Training data include the audio, some reference text, the ASR output and their alignment. We plan to make public at least part of the benchmark in view of a multilingual ASR benchmark for IWSLT 2014."
}
Markdown (Informal)
[Euronews: a multilingual speech corpus for ASR](https://preview.aclanthology.org/fix-sig-urls/L14-1546/) (Gretter, LREC 2014)
ACL
- Roberto Gretter. 2014. Euronews: a multilingual speech corpus for ASR. In Proceedings of the Ninth International Conference on Language Resources and Evaluation (LREC'14), pages 2635–2638, Reykjavik, Iceland. European Language Resources Association (ELRA).