@inproceedings{apaolaza-larraya-etal-2026-assessing,
title = "Assessing Logical Coherence of {LLM}s via Fine-Grained {NLI}",
author = "Apaolaza Larraya, Jon Felix and
Altuna, Bego{\~n}a and
Soroa, Aitor and
Lopez-Gazpio, Inigo",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.423/",
doi = "10.63317/4prei82n6ev9",
pages = "5431--5444",
abstract = "Natural Language Inference (NLI) is a long-standing probe of models' reasoning capabilities, yet it remains unclear how state-of-the-art systems represent and combine logical clauses in a way that supports robust generalization. We study directional effects in deductive NLI and introduce causal coherence, an evaluation paradigm that tests whether predictions remain consistent when the directionality of inference is reversed. Using fine-grained minimal-pair phrase data from PhrasIS, we evaluate encoder, decoder, and encoder{--}decoder transformers and analyze their behavior under both standard and manipulated settings. Our results show that models frequently fail to maintain logical stability when directionality varies, indicating shallow pattern matching rather than genuine clause composition. We formalize soft and hard causal coherence to disentangle directional consistency from correctness, and we provide an error analysis that highlights systematic failures involving semantic relations. Our findings suggest that deductive causal reasoning and coherence remain missing components in current transformer architectures, and that addressing them is necessary for reliable NLI."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="apaolaza-larraya-etal-2026-assessing">
<titleInfo>
<title>Assessing Logical Coherence of LLMs via Fine-Grained NLI</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jon</namePart>
<namePart type="given">Felix</namePart>
<namePart type="family">Apaolaza Larraya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Begoña</namePart>
<namePart type="family">Altuna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aitor</namePart>
<namePart type="family">Soroa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Inigo</namePart>
<namePart type="family">Lopez-Gazpio</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Natural Language Inference (NLI) is a long-standing probe of models’ reasoning capabilities, yet it remains unclear how state-of-the-art systems represent and combine logical clauses in a way that supports robust generalization. We study directional effects in deductive NLI and introduce causal coherence, an evaluation paradigm that tests whether predictions remain consistent when the directionality of inference is reversed. Using fine-grained minimal-pair phrase data from PhrasIS, we evaluate encoder, decoder, and encoder–decoder transformers and analyze their behavior under both standard and manipulated settings. Our results show that models frequently fail to maintain logical stability when directionality varies, indicating shallow pattern matching rather than genuine clause composition. We formalize soft and hard causal coherence to disentangle directional consistency from correctness, and we provide an error analysis that highlights systematic failures involving semantic relations. Our findings suggest that deductive causal reasoning and coherence remain missing components in current transformer architectures, and that addressing them is necessary for reliable NLI.</abstract>
<identifier type="citekey">apaolaza-larraya-etal-2026-assessing</identifier>
<identifier type="doi">10.63317/4prei82n6ev9</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.423/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>5431</start>
<end>5444</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Assessing Logical Coherence of LLMs via Fine-Grained NLI
%A Apaolaza Larraya, Jon Felix
%A Altuna, Begoña
%A Soroa, Aitor
%A Lopez-Gazpio, Inigo
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F apaolaza-larraya-etal-2026-assessing
%X Natural Language Inference (NLI) is a long-standing probe of models’ reasoning capabilities, yet it remains unclear how state-of-the-art systems represent and combine logical clauses in a way that supports robust generalization. We study directional effects in deductive NLI and introduce causal coherence, an evaluation paradigm that tests whether predictions remain consistent when the directionality of inference is reversed. Using fine-grained minimal-pair phrase data from PhrasIS, we evaluate encoder, decoder, and encoder–decoder transformers and analyze their behavior under both standard and manipulated settings. Our results show that models frequently fail to maintain logical stability when directionality varies, indicating shallow pattern matching rather than genuine clause composition. We formalize soft and hard causal coherence to disentangle directional consistency from correctness, and we provide an error analysis that highlights systematic failures involving semantic relations. Our findings suggest that deductive causal reasoning and coherence remain missing components in current transformer architectures, and that addressing them is necessary for reliable NLI.
%R 10.63317/4prei82n6ev9
%U https://aclanthology.org/2026.lrec-1.423/
%U https://doi.org/10.63317/4prei82n6ev9
%P 5431-5444
Markdown (Informal)
[Assessing Logical Coherence of LLMs via Fine-Grained NLI](https://aclanthology.org/2026.lrec-1.423/) (Apaolaza Larraya et al., LREC 2026)
ACL
- Jon Felix Apaolaza Larraya, Begoña Altuna, Aitor Soroa, and Inigo Lopez-Gazpio. 2026. Assessing Logical Coherence of LLMs via Fine-Grained NLI. In Proceedings of the Fifteenth Language Resources and Evaluation Conference, pages 5431–5444, Palma de Mallorca, Spain. ELRA Language Resource Association.