@inproceedings{perez-etal-2026-polyglotql,
title = "{P}olyglot{QL}: A Pipeline for Multilingual Text-to-{SPARQL} Dataset Generation",
author = "Perez, Julio and
Barth, Fabio and
Rehm, Georg",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/paragraph-normalization/2026.lrec-1.531/",
doi = "10.63317/5ow3k3fbz296",
pages = "6674--6684",
abstract = "We present PolyglotQL, an open-source ETL (Extract, Transform, Load) pipeline for systematically creating multilingual text-to-SPARQL datasets, along with an accompanying framework for evaluating text-to-SPARQL generation models. PolyglotQL provides an extensible and modular architecture that aggregates, normalizes, and augments heterogeneous question{--}SPARQL pairs from established text-to-SPARQL datasets. With this pipeline, we automatically construct a bilingual English{--}German dataset featuring contextualized entity and relationship mappings as well as automatically translated and aligned question pairs. We also conduct an empirical evaluation using two multilingual open large language models under two distinct contextualization settings. The results show consistent performance improvements when explicit grounding information is provided, highlighting the benefits of structured context in multilingual semantic parsing."
}Markdown (Informal)
[PolyglotQL: A Pipeline for Multilingual Text-to-SPARQL Dataset Generation](https://preview.aclanthology.org/paragraph-normalization/2026.lrec-1.531/) (Perez et al., LREC 2026)
ACL