@inproceedings{park-zubiaga-2026-better,
title = "Better Scores, Worse Grounding: Hidden Regressions after Fine-Tuning in Dialogue Fact Verification",
author = "Park, Hyunkyung and
Zubiaga, Arkaitz",
editor = "Choi, Jinho D. and
Chen, Yun-Nung and
Funakoshi, Kotaro and
Emami, Ali",
booktitle = "Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue",
month = aug,
year = "2026",
address = "Atlanta, Georgia, USA",
publisher = "Association for Computational Linguistics",
url = "https://preview.aclanthology.org/corrections-2026-07/2026.sigdial-1.51/",
pages = "720--737",
abstract = "In dialogue fact verification (DFV), responses often depend on prior turns for correct interpretation, yet systems are still judged mainly by aggregate benchmark scores. We study a hidden grounding regression: aggregate Macro-F1 improves after fine-tuning while previously correct, context-dependent pronoun cases become newly wrong and show stronger premise-side sensitivity than cases that remain correct. On three referent-annotated audit sets constructed from DialFact and FaithDial, we audit six encoder-only verifiers before and after matched source-specific fine-tuning through prediction-transition analysis and the Premise-Preference Score (PPS), a control-adjusted masking diagnostic. Fine-tuning improves Macro-F1 across the six-model/three-evaluation-set panel, yet newly regressed cases show stronger premise-side sensitivity than stable-correct cases under PPS in 17 of 18 evaluated comparisons, indicating that aggregate gains can conceal regressions on a controlled dialogue-grounding audit."
}Markdown (Informal)
[Better Scores, Worse Grounding: Hidden Regressions after Fine-Tuning in Dialogue Fact Verification](https://preview.aclanthology.org/corrections-2026-07/2026.sigdial-1.51/) (Park & Zubiaga, SIGDIAL 2026)
ACL