@inproceedings{yeshpanov-2026-using,
title = "Using Songs to Improve {K}azakh Automatic Speech Recognition",
author = "Yeshpanov, Rustem",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/paragraph-normalization/2026.lrec-1.431/",
doi = "10.63317/5hqonmz5roum",
pages = "5528--5537",
abstract = "Developing automatic speech recognition (ASR) systems for low-resource languages is hindered by the scarcity of transcribed corpora. This proof-of-concept study explores songs as an unconventional yet promising data source for Kazakh ASR. We curate a dataset of 3,013 audio-text pairs (about 4.5 hours) from 195 songs by 36 artists, segmented at the lyric-line level. Using Whisper as the base recogniser, we fine-tune models under seven training scenarios involving Songs, Common Voice Corpus (CVC), and FLEURS, and evaluate them on three benchmarks: CVC, FLEURS, and Kazakh Speech Corpus 2 (KSC2). Results show that song-based fine-tuning improves performance over zero-shot baselines. For instance, Whisper Large-V3 Turbo trained on a mixture of Songs, CVC, and FLEURS achieves 27.6{\%} normalised WER on CVC and 11.8{\%} on FLEURS, while halving the error on KSC2 (39.3{\%} vs. 81.2{\%}) relative to the zero-shot model. Although these gains remain below those of models trained on the 1,100-hour KSC2 corpus, they demonstrate that even modest song-speech mixtures can yield meaningful adaptation improvements in low-resource ASR. The dataset is released on Hugging Face for research purposes under a gated, non-commercial licence."
}Markdown (Informal)
[Using Songs to Improve Kazakh Automatic Speech Recognition](https://preview.aclanthology.org/paragraph-normalization/2026.lrec-1.431/) (Yeshpanov, LREC 2026)
ACL