@inproceedings{raihan-etal-2026-tigercoder,
title = "{T}iger{C}oder: A Novel Suite of {LLM}s for Code Generation in {B}angla",
author = "Raihan, Nishat and
Anastasopoulos, Antonios and
Zampieri, Marcos",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/ingest-nlpsi/2026.lrec-1.238/",
doi = "10.63317/5nampb63np3m",
pages = "3044--3054",
abstract = "Despite being the 5th most spoken language, Bangla remains underrepresented in Large Language Models (LLMs), particularly for code generation. This primarily stems from the scarcity of high-quality data to pre-train and/or finetune such models. Hence, we introduce the first dedicated family of Code LLMs for Bangla (1B {\&} 9B). We offer three major contributions: (1) a comprehensive Bangla code instruction datasets for programming domain adaptation; (2) MBPP-Bangla, an evaluation benchmark for Bangla code generation; and (3) the TigerCoder-family of Code LLMs, achieving significant {\textasciitilde}11-18{\%} performance gains at Pass@1 over existing multilingual and general-purpose Bangla LLMs. Our findings show that curated, high-quality datasets can overcome limitations of smaller models for low-resource languages."
}Markdown (Informal)
[TigerCoder: A Novel Suite of LLMs for Code Generation in Bangla](https://preview.aclanthology.org/ingest-nlpsi/2026.lrec-1.238/) (Raihan et al., LREC 2026)
ACL