@inproceedings{nguyen-quan-2026-works,
title = "Which Works Best for {V}ietnamese? {A} Practical Study of Information Retrieval Methods across Domains",
author = "Nguyen, Long S. T. and
Quan, Tho T.",
editor = "Demberg, Vera and
Inui, Kentaro and
Marquez, Llu{\'i}s",
booktitle = "Findings of the {A}ssociation for {C}omputational {L}inguistics: {EACL} 2026",
month = mar,
year = "2026",
address = "Rabat, Morocco",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/issues-pwc/2026.findings-eacl.110/",
doi = "10.18653/v1/2026.findings-eacl.110",
pages = "2098--2119",
ISBN = "979-8-89176-386-9",
abstract = "Large Language Models (LLMs) depend on retrieval for factual grounding in Retrieval-Augmented Generation (RAG), placing Information Retrieval (IR) at the core of modern Question Answering (QA) systems. While lexical, dense, and hybrid paradigms have been extensively benchmarked in English, their relative effectiveness for Vietnamese remains insufficiently characterized, especially under realistic multi-domain settings. Existing studies are typically confined to single domains or curated datasets, limiting cross-domain comparability and obscuring paradigm-level trade-offs. We introduce the first domain-normalized, multi-domain benchmark for Vietnamese IR under a unified and reproducible evaluation protocol, spanning six domains and ten datasets across education, legal, healthcare, customer support, lifestyle reviews, and open-domain knowledge. We evaluate lexical, neural-sparse, late-interaction, dense, and hybrid paradigms across diverse Vietnamese-specific and multilingual embedding backbones, and release two QA datasets, EduCoQA and CSConDa, constructed from authentic counseling and customer-service interactions. Beyond reporting benchmark performance, we derive systematic insights into lexical{--}semantic hybridization, specialization versus robustness trade-offs, and the limited predictive value of model scale for retrieval effectiveness. All datasets and evaluation scripts are publicly available at \url{https://github.com/longstnguyen/ViRE}."
}Markdown (Informal)
[Which Works Best for Vietnamese? A Practical Study of Information Retrieval Methods across Domains](https://preview.aclanthology.org/issues-pwc/2026.findings-eacl.110/) (Nguyen & Quan, Findings 2026)
ACL