@inproceedings{park-zubiaga-2026-better,
title = "Better Scores, Worse Grounding: Hidden Regressions after Fine-Tuning in Dialogue Fact Verification",
author = "Park, Hyunkyung and
Zubiaga, Arkaitz",
editor = "Choi, Jinho D. and
Chen, Yun-Nung and
Funakoshi, Kotaro and
Emami, Ali",
booktitle = "Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue",
month = aug,
year = "2026",
address = "Atlanta, Georgia, USA",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.sigdial-1.51/",
pages = "720--737",
abstract = "In dialogue fact verification (DFV), responses often depend on prior turns for correct interpretation, yet systems are still judged mainly by aggregate benchmark scores. We study a hidden grounding regression: aggregate Macro-F1 improves after fine-tuning while previously correct, context-dependent pronoun cases become newly wrong and show stronger premise-side sensitivity than cases that remain correct. On three referent-annotated audit sets constructed from DialFact and FaithDial, we audit six encoder-only verifiers before and after matched source-specific fine-tuning through prediction-transition analysis and the Premise-Preference Score (PPS), a control-adjusted masking diagnostic. Fine-tuning improves Macro-F1 across the six-model/three-evaluation-set panel, yet newly regressed cases show stronger premise-side sensitivity than stable-correct cases under PPS in 17 of 18 evaluated comparisons, indicating that aggregate gains can conceal regressions on a controlled dialogue-grounding audit."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="park-zubiaga-2026-better">
<titleInfo>
<title>Better Scores, Worse Grounding: Hidden Regressions after Fine-Tuning in Dialogue Fact Verification</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hyunkyung</namePart>
<namePart type="family">Park</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Arkaitz</namePart>
<namePart type="family">Zubiaga</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-08</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jinho</namePart>
<namePart type="given">D</namePart>
<namePart type="family">Choi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yun-Nung</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kotaro</namePart>
<namePart type="family">Funakoshi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ali</namePart>
<namePart type="family">Emami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Atlanta, Georgia, USA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>In dialogue fact verification (DFV), responses often depend on prior turns for correct interpretation, yet systems are still judged mainly by aggregate benchmark scores. We study a hidden grounding regression: aggregate Macro-F1 improves after fine-tuning while previously correct, context-dependent pronoun cases become newly wrong and show stronger premise-side sensitivity than cases that remain correct. On three referent-annotated audit sets constructed from DialFact and FaithDial, we audit six encoder-only verifiers before and after matched source-specific fine-tuning through prediction-transition analysis and the Premise-Preference Score (PPS), a control-adjusted masking diagnostic. Fine-tuning improves Macro-F1 across the six-model/three-evaluation-set panel, yet newly regressed cases show stronger premise-side sensitivity than stable-correct cases under PPS in 17 of 18 evaluated comparisons, indicating that aggregate gains can conceal regressions on a controlled dialogue-grounding audit.</abstract>
<identifier type="citekey">park-zubiaga-2026-better</identifier>
<location>
<url>https://aclanthology.org/2026.sigdial-1.51/</url>
</location>
<part>
<date>2026-08</date>
<extent unit="page">
<start>720</start>
<end>737</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings %T Better Scores, Worse Grounding: Hidden Regressions after Fine-Tuning in Dialogue Fact Verification %A Park, Hyunkyung %A Zubiaga, Arkaitz %Y Choi, Jinho D. %Y Chen, Yun-Nung %Y Funakoshi, Kotaro %Y Emami, Ali %S Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue %D 2026 %8 August %I Association for Computational Linguistics %C Atlanta, Georgia, USA %F park-zubiaga-2026-better %X In dialogue fact verification (DFV), responses often depend on prior turns for correct interpretation, yet systems are still judged mainly by aggregate benchmark scores. We study a hidden grounding regression: aggregate Macro-F1 improves after fine-tuning while previously correct, context-dependent pronoun cases become newly wrong and show stronger premise-side sensitivity than cases that remain correct. On three referent-annotated audit sets constructed from DialFact and FaithDial, we audit six encoder-only verifiers before and after matched source-specific fine-tuning through prediction-transition analysis and the Premise-Preference Score (PPS), a control-adjusted masking diagnostic. Fine-tuning improves Macro-F1 across the six-model/three-evaluation-set panel, yet newly regressed cases show stronger premise-side sensitivity than stable-correct cases under PPS in 17 of 18 evaluated comparisons, indicating that aggregate gains can conceal regressions on a controlled dialogue-grounding audit. %U https://aclanthology.org/2026.sigdial-1.51/ %P 720-737
Markdown (Informal)
[Better Scores, Worse Grounding: Hidden Regressions after Fine-Tuning in Dialogue Fact Verification](https://aclanthology.org/2026.sigdial-1.51/) (Park & Zubiaga, SIGDIAL 2026)