@inproceedings{vladu-etal-2026-building,
title = "Building Collaborative Speech Corpora for Low-Resource Languages: The {G}alician Dataset in Mozilla Common Voice",
author = "Vladu, Adina Ioana and
Fern{\'a}ndez Rei, Elisa and
P{\'e}rez Lago, Mar{\'i}a",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/cawl-year/2026.lrec-1.135/",
doi = "10.63317/4gd79cq3cump",
pages = "1710--1720",
abstract = "This paper presents the methodology and outcomes of building collaborative speech corpora in Mozilla Common Voice (MCV), focusing on the Galician case within Proxecto N{\'o}s. We describe the organization of voice collection campaigns {--}on-site events, student participation, Validat{\'o}n marathons, and corporate collaboration{--} and analyze the results in MCV v22.0. While the dataset has achieved a modest scale, major gaps remain in metadata completeness and dialectal tagging, with implications for ASR performance. Drawing on our experience, we highlight effective strategies for engagement, such as transparent communication, cultural identification, and user-friendly tools. We conclude with lessons learnt for improving data representativeness, participant retention, and ethical governance. The observations are specific to the Galician case study but may inform similar efforts in other lesser-resourced languages."
}Markdown (Informal)
[Building Collaborative Speech Corpora for Low-Resource Languages: The Galician Dataset in Mozilla Common Voice](https://preview.aclanthology.org/cawl-year/2026.lrec-1.135/) (Vladu et al., LREC 2026)
ACL