@inproceedings{ali-etal-2026-sdquad,
title = "{S}d{Q}u{AD}: A Large Benchmark Question Answering Dataset for Low-resource {S}indhi Language",
author = "Ali, Wazir and
Shaikh, Muhammad Rafay and
Ali, Nadia and
Rehman, Amar",
editor = "Morger, Felix and
Ilinykh, Nikolai and
Scalvini, Barbara and
Dobnik, Simon and
Dann{\'e}lls, Dana",
booktitle = "Proceedings of the Fourth Workshop on the Role of Resources in the Age of Large Language Models ({RESOURCEFUL} 2026)",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/cawl-year/2026.resourceful-4.6/",
doi = "10.63317/3dhhfxeoztgo",
pages = "55--61",
abstract = "Question answering (QA) datasets are crucial for developing and evaluating monolingual and multilingual language models, yet low-resource languages like Sindhi lack open-source QA resources. We introduce SdQuAD, a novel open-source textual QA dataset for the low-resource Sindhi language, comprising 15,000 QA pairs meticulously annotated by native speakers using the Label Studio platform. Sourced from diverse domains, including news, history, science, geography, business, and tourism, SdQuAD supports both extractive and abstractive QA tasks while capturing Sindhi{'}s linguistic and topical diversity. We assess annotation quality using span-level agreement and evaluate extractive performance with Exact Match (EM), F1 score, and a TF-IDF baseline. Additionally, we fine-tune mBERT, XLM-R, and mT5 models on SdQuAD, benchmarking their performance to demonstrate the dataset{'}s utility."
}Markdown (Informal)
[SdQuAD: A Large Benchmark Question Answering Dataset for Low-resource Sindhi Language](https://preview.aclanthology.org/cawl-year/2026.resourceful-4.6/) (Ali et al., RESOURCEFUL 2026)
ACL