@inproceedings{parikh-solorio-2021-normalization,
title = "Normalization and Back-Transliteration for Code-Switched Data",
author = "Parikh, Dwija and
Solorio, Thamar",
editor = "Solorio, Thamar and
Chen, Shuguang and
Black, Alan W. and
Diab, Mona and
Sitaram, Sunayana and
Soto, Victor and
Yilmaz, Emre and
Srinivasan, Anirudh",
booktitle = "Proceedings of the Fifth Workshop on Computational Approaches to Linguistic Code-Switching",
month = jun,
year = "2021",
address = "Online",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/fix-sig-urls/2021.calcs-1.15/",
doi = "10.18653/v1/2021.calcs-1.15",
pages = "119--124",
abstract = "Code-switching is an omnipresent phenomenon in multilingual communities all around the world but remains a challenge for NLP systems due to the lack of proper data and processing techniques. Hindi-English code-switched text on social media is often transliterated to the Roman script which prevents from utilizing monolingual resources available in the native Devanagari script. In this paper, we propose a method to normalize and back-transliterate code-switched Hindi-English text. In addition, we present a grapheme-to-phoneme (G2P) conversion technique for romanized Hindi data. We also release a dataset of script-corrected Hindi-English code-switched sentences labeled for the named entity recognition and part-of-speech tagging tasks to facilitate further research."
}
Markdown (Informal)
[Normalization and Back-Transliteration for Code-Switched Data](https://preview.aclanthology.org/fix-sig-urls/2021.calcs-1.15/) (Parikh & Solorio, CALCS 2021)
ACL