@inproceedings{davidsson-harmouch-2026-end,
title = "End-to-End Graph Retrieval Pipeline for Specialized Domains",
author = "Davidsson, Haraldur and
Harmouch, Hazar",
editor = "S{\'e}rasset, Gilles and
Gkirtzou, Katerina and
Cochez, Michael and
Kalo, Jan-Christoph",
booktitle = "Proceedings of the Knowledge Graphs and Large Language Models Workshop ({KG}-{LLM}) @ {LREC}26",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/paragraph-normalization/2026.kallm-1.16/",
doi = "10.63317/3xjhui2zyiws",
pages = "155--165",
abstract = "We present an end-to-end pipeline for constructing a domain-specific knowledge graph from instructional text using Large Language Model assisted extraction. Applied to the Icelandic Riding Levels, a 602 pages training corpus for riders of the Icelandic Horse, the pipeline produces a hyper-relational knowledge graph of 9,382 nodes and 16,423 edges, where schema-constrained qualifiers preserve the conditional and procedural context that standard triples discard. To evaluate the resulting graph, we introduce the first expert validated question answering benchmark for this domain: 252 questions across four reasoning categories. Comparing Graph-, Text-, and Hybrid-retrieval augmented generation methods, we find that Text-based achieves the highest overall accuracy, but that Graph-based provides the only correct answer for a subset of queries, particularly where the corpus contains competing values for the same fact. A failure analysis traces the majority of Graph-based retrieval errors to context dilution at high-degree hub nodes, an algorithmic limitation in graph traversal. We discuss implications for adaptive retrieval strategies that route queries to the appropriate modality."
}Markdown (Informal)
[End-to-End Graph Retrieval Pipeline for Specialized Domains](https://preview.aclanthology.org/paragraph-normalization/2026.kallm-1.16/) (Davidsson & Harmouch, KaLLM 2026)
ACL