@inproceedings{nyalang-2026-ne-lid,
title = "{NE}-{LID}: A Fast and Accurate Language Identification System for {N}ortheast {I}ndian Languages",
author = "Nyalang, Badal",
editor = "Jha, Girish Nath and
Bali, Kalika and
L, Sobha and
Kumar, Devendr",
booktitle = "Proceedings of the 8th Workshop on {I}ndian Language Data: Resources and Evaluation",
month = may,
year = "2026",
address = "Palma, Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/revision-workflow/2026.wildre-1.14/",
doi = "10.63317/2q69fiyg73p6",
pages = "104--108",
abstract = "Language identification (LID) is crucial for natural language processing systems, yet Northeast Indian languages remain severely underserved by existing multilingual LID models. We present NE-LID, a fast and accurate language identification system specifically designed for eleven languages of Northeast India. Built using character n-gram features with fastText, NE-LID achieves 99.09{\%} accuracy on a balanced test set, significantly outperforming existing multilingual systems including GlotLID (73.12{\%}), OpenLID (42.03{\%}), IndicLID (39.30{\%}), and LangDetect (24.33{\%}). Our model processes predictions in 0.084 milliseconds on average, enabling real-time applications. We demonstrate that character-level modeling outperforms transformer-based approaches for script-diverse, low-resource languages"
}Markdown (Informal)
[NE-LID: A Fast and Accurate Language Identification System for Northeast Indian Languages](https://preview.aclanthology.org/revision-workflow/2026.wildre-1.14/) (Nyalang, WILDRE 2026)
ACL