@inproceedings{ren-etal-2026-detecting,
title = "Detecting Hallucinations in Authentic {LLM}{--}Human Interactions",
author = "Ren, Yujie and
Gruhlke, Niklas and
Lauscher, Anne",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://preview.aclanthology.org/test-year-match/2026.lrec-1.475/",
doi = "10.63317/5pykihiz52tk",
pages = "5981--5995",
abstract = "As large language models (LLMs) are increasingly applied in sensitive domains such as medicine and law, hallucination detection has become a critical task. Although numerous benchmarks have been proposed to advance research in this area, most of them are artificially constructed{--}{--}either through deliberate hallucination induction or simulated interactions{--}{--}rather than derived from genuine LLM{--}human dialogues. Consequently, these benchmarks fail to fully capture the characteristics of hallucinations that occur in real-world usage. To address this limitation, we introduce AuthenHallu, the first hallucination detection benchmark built entirely from authentic LLM{--}human interactions. For AuthenHallu, we select and annotate samples from genuine LLM{--}human dialogues, thereby providing a faithful reflection of how LLMs hallucinate in everyday user interactions. Statistical analysis shows that hallucinations occur in 31.4{\%} of the query{--}response pairs in our benchmark, and this proportion increases dramatically to 60.0{\%} in challenging domains such as `Math {\&} Number Problems'. Furthermore, we explore the potential of using vanilla LLMs themselves as hallucination detectors and find that, despite some promise, their current performance remains insufficient in real-world scenarios. The data and code are publicly available at \url{https://github.com/TAI-HAMBURG/AuthenHallu}."
}Markdown (Informal)
[Detecting Hallucinations in Authentic LLM–Human Interactions](https://preview.aclanthology.org/test-year-match/2026.lrec-1.475/) (Ren et al., LREC 2026)
ACL