@inproceedings{hoefels-2026-infact,
title = "{I}n{FACT}: Benchmarking {LLM} Explanations Against Institutional Reasoning for Deliberation-Aware Fact-Checking",
author = "Hoefels, Diana Constantina",
editor = "Anastasiou, Lucas and
Boland, Katarina and
Liddo, Anna De and
Falk, Neele and
Hautli-Janisz, Annette and
Lapesa, Gabriella and
Romberg, Julia",
booktitle = "Proceedings of The 2nd Workshop on Language-driven Deliberation Technology",
month = may,
year = "2026",
address = "Mallorca, Spain",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.delite-1.3/",
doi = "10.63317/5bp93rjt27hy",
pages = "18--28",
abstract = "Explainability in deliberation-support NLP is usually evaluated through post-hoc rationales or model-internal attribution methods, and only rarely against explicit institutional reasoning procedures. We introduce , a Romanian corpus of professional fact-checking reports that preserves the workflow of editorial epistemic arbitration, namely claim articulation, contextualisation, verification scope, evidence-based verification narrative, and calibrated conclusion. contains 789 raw reports from \textit{factual.ro} and a processed benchmark release of 788 instances after removal of a singleton non-standard verdict label. Beyond six-way verdict prediction, we position as a benchmark for LLM explanation alignment, where models must generate short explanations that can be compared directly to gold institutional reasoning. We evaluate primarily with instruction-tuned LLMs, reporting full-corpus experiments for open-weight models and a matched pilot comparison with GPT-4 Turbo. The resulting evidence shows that verdict prediction and institutional explanation alignment are not the same capability: models that improve verdict accuracy do not necessarily preserve institutional calibration or produce explanations that align with professional verification narratives. These results support the central claim of the paper, namely that measures not only whether a model reaches a verdict, but also whether it does so in a manner that resembles documented public reasoning."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hoefels-2026-infact">
<titleInfo>
<title>InFACT: Benchmarking LLM Explanations Against Institutional Reasoning for Deliberation-Aware Fact-Checking</title>
</titleInfo>
<name type="personal">
<namePart type="given">Diana</namePart>
<namePart type="given">Constantina</namePart>
<namePart type="family">Hoefels</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of The 2nd Workshop on Language-driven Deliberation Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Lucas</namePart>
<namePart type="family">Anastasiou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Katarina</namePart>
<namePart type="family">Boland</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anna</namePart>
<namePart type="given">De</namePart>
<namePart type="family">Liddo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Neele</namePart>
<namePart type="family">Falk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Annette</namePart>
<namePart type="family">Hautli-Janisz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gabriella</namePart>
<namePart type="family">Lapesa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Julia</namePart>
<namePart type="family">Romberg</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Explainability in deliberation-support NLP is usually evaluated through post-hoc rationales or model-internal attribution methods, and only rarely against explicit institutional reasoning procedures. We introduce , a Romanian corpus of professional fact-checking reports that preserves the workflow of editorial epistemic arbitration, namely claim articulation, contextualisation, verification scope, evidence-based verification narrative, and calibrated conclusion. contains 789 raw reports from factual.ro and a processed benchmark release of 788 instances after removal of a singleton non-standard verdict label. Beyond six-way verdict prediction, we position as a benchmark for LLM explanation alignment, where models must generate short explanations that can be compared directly to gold institutional reasoning. We evaluate primarily with instruction-tuned LLMs, reporting full-corpus experiments for open-weight models and a matched pilot comparison with GPT-4 Turbo. The resulting evidence shows that verdict prediction and institutional explanation alignment are not the same capability: models that improve verdict accuracy do not necessarily preserve institutional calibration or produce explanations that align with professional verification narratives. These results support the central claim of the paper, namely that measures not only whether a model reaches a verdict, but also whether it does so in a manner that resembles documented public reasoning.</abstract>
<identifier type="citekey">hoefels-2026-infact</identifier>
<identifier type="doi">10.63317/5bp93rjt27hy</identifier>
<location>
<url>https://aclanthology.org/2026.delite-1.3/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>18</start>
<end>28</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T InFACT: Benchmarking LLM Explanations Against Institutional Reasoning for Deliberation-Aware Fact-Checking
%A Hoefels, Diana Constantina
%Y Anastasiou, Lucas
%Y Boland, Katarina
%Y Liddo, Anna De
%Y Falk, Neele
%Y Hautli-Janisz, Annette
%Y Lapesa, Gabriella
%Y Romberg, Julia
%S Proceedings of The 2nd Workshop on Language-driven Deliberation Technology
%D 2026
%8 May
%I Association for Computational Linguistics
%C Mallorca, Spain
%F hoefels-2026-infact
%X Explainability in deliberation-support NLP is usually evaluated through post-hoc rationales or model-internal attribution methods, and only rarely against explicit institutional reasoning procedures. We introduce , a Romanian corpus of professional fact-checking reports that preserves the workflow of editorial epistemic arbitration, namely claim articulation, contextualisation, verification scope, evidence-based verification narrative, and calibrated conclusion. contains 789 raw reports from factual.ro and a processed benchmark release of 788 instances after removal of a singleton non-standard verdict label. Beyond six-way verdict prediction, we position as a benchmark for LLM explanation alignment, where models must generate short explanations that can be compared directly to gold institutional reasoning. We evaluate primarily with instruction-tuned LLMs, reporting full-corpus experiments for open-weight models and a matched pilot comparison with GPT-4 Turbo. The resulting evidence shows that verdict prediction and institutional explanation alignment are not the same capability: models that improve verdict accuracy do not necessarily preserve institutional calibration or produce explanations that align with professional verification narratives. These results support the central claim of the paper, namely that measures not only whether a model reaches a verdict, but also whether it does so in a manner that resembles documented public reasoning.
%R 10.63317/5bp93rjt27hy
%U https://aclanthology.org/2026.delite-1.3/
%U https://doi.org/10.63317/5bp93rjt27hy
%P 18-28
Markdown (Informal)
[InFACT: Benchmarking LLM Explanations Against Institutional Reasoning for Deliberation-Aware Fact-Checking](https://aclanthology.org/2026.delite-1.3/) (Hoefels, DELITE 2026)
ACL