@inproceedings{hammerla-mehler-2026-negation,
title = "Negation in Reasoning Traces: Interpretable Signals of Correctness and Provenance",
author = "Hammerla, Leon Lukas and
Mehler, Alexander",
editor = "Yanaka, Hitomi and
Abzianidze, Lasha",
booktitle = "Proceedings of the 6th Workshop on Natural Language Meets Logic and Machine Learning ({NALOMA})",
month = aug,
year = "2026",
address = "Prague, Czechia",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.naloma-1.4/",
pages = "19--39",
ISBN = "979-8-89176-389-0",
abstract = "Chain-of-thought (CoT) reasoning is widely used in large language models (LLMs), but the resulting reasoning traces remain underexplored.We study these traces through the lens of discourse-level negation.Specifically, we distinguish between corrective negation, which rejects a prior reasoning step, and refining negation, which narrows or qualifies it, and introduce metrics to quantify their use in human- and LLM-authored reasoning traces.Across multiple benchmarks, we find that negation occurs much more frequently in intermediate reasoning traces than in final response texts.We then test whether negation-based features provide predictive and descriptive signal for correctness, model identity, and human-vs.-LLM authorship.For correctness prediction, negation-based features consistently outperform simple structural baselines and in several settings add complementary signal to embedding-based representations, although embeddings remain stronger overall.In a controlled comparison on correct human and LLM traces from the same dataset, our strongest results arise in human-vs.-LLM classification, where negation features outperform both structural and embedding baselines.Overall, these findings position discourse-level negation as an interpretable feature for reasoning-trace analysis, with especially strong utility for provenance-related classification and modest but consistent value for correctness prediction."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hammerla-mehler-2026-negation">
<titleInfo>
<title>Negation in Reasoning Traces: Interpretable Signals of Correctness and Provenance</title>
</titleInfo>
<name type="personal">
<namePart type="given">Leon</namePart>
<namePart type="given">Lukas</namePart>
<namePart type="family">Hammerla</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alexander</namePart>
<namePart type="family">Mehler</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-08</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 6th Workshop on Natural Language Meets Logic and Machine Learning (NALOMA)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hitomi</namePart>
<namePart type="family">Yanaka</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lasha</namePart>
<namePart type="family">Abzianidze</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Prague, Czechia</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-89176-389-0</identifier>
</relatedItem>
<abstract>Chain-of-thought (CoT) reasoning is widely used in large language models (LLMs), but the resulting reasoning traces remain underexplored.We study these traces through the lens of discourse-level negation.Specifically, we distinguish between corrective negation, which rejects a prior reasoning step, and refining negation, which narrows or qualifies it, and introduce metrics to quantify their use in human- and LLM-authored reasoning traces.Across multiple benchmarks, we find that negation occurs much more frequently in intermediate reasoning traces than in final response texts.We then test whether negation-based features provide predictive and descriptive signal for correctness, model identity, and human-vs.-LLM authorship.For correctness prediction, negation-based features consistently outperform simple structural baselines and in several settings add complementary signal to embedding-based representations, although embeddings remain stronger overall.In a controlled comparison on correct human and LLM traces from the same dataset, our strongest results arise in human-vs.-LLM classification, where negation features outperform both structural and embedding baselines.Overall, these findings position discourse-level negation as an interpretable feature for reasoning-trace analysis, with especially strong utility for provenance-related classification and modest but consistent value for correctness prediction.</abstract>
<identifier type="citekey">hammerla-mehler-2026-negation</identifier>
<location>
<url>https://aclanthology.org/2026.naloma-1.4/</url>
</location>
<part>
<date>2026-08</date>
<extent unit="page">
<start>19</start>
<end>39</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Negation in Reasoning Traces: Interpretable Signals of Correctness and Provenance
%A Hammerla, Leon Lukas
%A Mehler, Alexander
%Y Yanaka, Hitomi
%Y Abzianidze, Lasha
%S Proceedings of the 6th Workshop on Natural Language Meets Logic and Machine Learning (NALOMA)
%D 2026
%8 August
%I Association for Computational Linguistics
%C Prague, Czechia
%@ 979-8-89176-389-0
%F hammerla-mehler-2026-negation
%X Chain-of-thought (CoT) reasoning is widely used in large language models (LLMs), but the resulting reasoning traces remain underexplored.We study these traces through the lens of discourse-level negation.Specifically, we distinguish between corrective negation, which rejects a prior reasoning step, and refining negation, which narrows or qualifies it, and introduce metrics to quantify their use in human- and LLM-authored reasoning traces.Across multiple benchmarks, we find that negation occurs much more frequently in intermediate reasoning traces than in final response texts.We then test whether negation-based features provide predictive and descriptive signal for correctness, model identity, and human-vs.-LLM authorship.For correctness prediction, negation-based features consistently outperform simple structural baselines and in several settings add complementary signal to embedding-based representations, although embeddings remain stronger overall.In a controlled comparison on correct human and LLM traces from the same dataset, our strongest results arise in human-vs.-LLM classification, where negation features outperform both structural and embedding baselines.Overall, these findings position discourse-level negation as an interpretable feature for reasoning-trace analysis, with especially strong utility for provenance-related classification and modest but consistent value for correctness prediction.
%U https://aclanthology.org/2026.naloma-1.4/
%P 19-39
Markdown (Informal)
[Negation in Reasoning Traces: Interpretable Signals of Correctness and Provenance](https://aclanthology.org/2026.naloma-1.4/) (Hammerla & Mehler, NALOMA 2026)
ACL