@inproceedings{halabi-wald-2016-phonetic,
title = "Phonetic Inventory for an {A}rabic Speech Corpus",
author = "Halabi, Nawar and
Wald, Mike",
editor = "Calzolari, Nicoletta and
Choukri, Khalid and
Declerck, Thierry and
Goggi, Sara and
Grobelnik, Marko and
Maegaard, Bente and
Mariani, Joseph and
Mazo, Helene and
Moreno, Asuncion and
Odijk, Jan and
Piperidis, Stelios",
booktitle = "Proceedings of the Tenth International Conference on Language Resources and Evaluation ({LREC}`16)",
month = may,
year = "2016",
address = "Portoro{\v{z}}, Slovenia",
publisher = "European Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/add-emnlp-2024-awards/L16-1116/",
pages = "734--738",
abstract = "Corpus design for speech synthesis is a well-researched topic in languages such as English compared to Modern Standard Arabic, and there is a tendency to focus on methods to automatically generate the orthographic transcript to be recorded (usually greedy methods). In this work, a study of Modern Standard Arabic (MSA) phonetics and phonology is conducted in order to create criteria for a greedy method to create a speech corpus transcript for recording. The size of the dataset is reduced a number of times using these optimisation methods with different parameters to yield a much smaller dataset with identical phonetic coverage than before the reduction, and this output transcript is chosen for recording. This is part of a larger work to create a completely annotated and segmented speech corpus for MSA."
}
Markdown (Informal)
[Phonetic Inventory for an Arabic Speech Corpus](https://preview.aclanthology.org/add-emnlp-2024-awards/L16-1116/) (Halabi & Wald, LREC 2016)
ACL
- Nawar Halabi and Mike Wald. 2016. Phonetic Inventory for an Arabic Speech Corpus. In Proceedings of the Tenth International Conference on Language Resources and Evaluation (LREC'16), pages 734–738, Portorož, Slovenia. European Language Resources Association (ELRA).