@inproceedings{do-bloem-2026-well,
title = "How Well Do Large Language Models Reason in Under-Resourced Languages? Evidence from {V}ietnamese",
author = "Do, Tuan Anh and
Bloem, Jelke",
editor = "Ojha, Atul Kr. and
Sakti, Sakriani and
Soria, Claudia and
Melero, Maite and
McCrae, John P. and
Lignos, Constantine and
Liu, Chao-Hong and
Claramunt, German Rigau and
Rehm, Georg",
booktitle = "Proceedings of the {SIGUL} 2026 Joint Workshop with {ELE}, {EURALI}, and {DCLRL}: Towards Inclusivity and Equality: Language Resources and Technologies for Under-Resourced and Endangered Languages",
month = may,
year = "2026",
address = "Palma, Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/test-year-match/2026.sigul-1.1/",
doi = "10.63317/2a43bkurpywk",
pages = "1--18",
abstract = "Despite advancements in Large Language Models, reasoning benchmarks remain centered on high-resource languages, leaving languages like Vietnamese under-evaluated. In this study, we aim to address this gap by evaluating four models: PhoGPT (native), Vistral and VBD-Llama (adapted), and Llama-2 (English-centric), on commonsense reasoning and arithmetic reasoning. As Vietnamese benchmarks for these tasks are lacking, we adapt two analogy datasets from English to Vietnamese and construct two sequence datasets, ensuring a range of structural complexity and difficulty levels. We evaluate diverse prompting strategies, including Chain-of-Thought, role-playing guidance, cross-lingual prompting, and few-shot learning. Our results reveal a baseline proficiency in analogical and arithmetic reasoning among the models, with Vistral and Llama-2 outperforming other models in multiple tasks. The effects of Chain-of-Thought and contextual guidance are limited in Vietnamese, while cross-lingual prompting and few-shot learning show promising performance improvements. The findings underscore the feasibility of adapting benchmarks to less-resourced languages and provide insights into strengths and weaknesses in the performance of Vietnamese LLMs, suggesting directions for model improvements."
}Markdown (Informal)
[How Well Do Large Language Models Reason in Under-Resourced Languages? Evidence from Vietnamese](https://preview.aclanthology.org/test-year-match/2026.sigul-1.1/) (Do & Bloem, SIGUL-EURALI-DCLRL 2026)
ACL