@article{nadkarni-etal-2026-rewriting,
title = "Rewriting History: A Recipe for Interventional Analyses to Study Data Effects on Model Behavior",
author = "Nadkarni, Rahul and
Elazar, Yanai and
Gonen, Hila and
Smith, Noah A.",
journal = "Transactions of the Association for Computational Linguistics",
volume = "14",
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.tacl-1.70/",
doi = "10.1162/tacl.a.740",
pages = "1562--1588",
abstract = "We present an experimental recipe for studying the relationship between training data and language model (LM) behavior. We outline steps for intervening on data batches {--} i.e., ``rewriting history'' {--} and then retraining model checkpoints over that data to test hypotheses relating data to behavior. Our intervention recipe{'}s stages are (1) selecting evaluation items from a benchmark that measures model behavior, (2) matching relevant documents to those items, and (3) modifying those documents before retraining and measuring the effects. We demonstrate the utility of our recipe through case studies on factual knowledge acquisition and gender bias in LMs, using both cooccurrence statistics and information retrieval methods to identify documents that might contribute to model behavior. Our results supplement past observational analyses that link cooccurrence to model behavior, while demonstrating that extant methods for identifying relevant training documents do not fully explain an LM{'}s abilities and biases. Researchers can follow the recipe to test further hypotheses about how training data affects model behavior. Our code is made publicly available to promote future work.1"
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="nadkarni-etal-2026-rewriting">
<titleInfo>
<title>Rewriting History: A Recipe for Interventional Analyses to Study Data Effects on Model Behavior</title>
</titleInfo>
<name type="personal">
<namePart type="given">Rahul</namePart>
<namePart type="family">Nadkarni</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yanai</namePart>
<namePart type="family">Elazar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hila</namePart>
<namePart type="family">Gonen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Noah</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Smith</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>We present an experimental recipe for studying the relationship between training data and language model (LM) behavior. We outline steps for intervening on data batches – i.e., “rewriting history” – and then retraining model checkpoints over that data to test hypotheses relating data to behavior. Our intervention recipe’s stages are (1) selecting evaluation items from a benchmark that measures model behavior, (2) matching relevant documents to those items, and (3) modifying those documents before retraining and measuring the effects. We demonstrate the utility of our recipe through case studies on factual knowledge acquisition and gender bias in LMs, using both cooccurrence statistics and information retrieval methods to identify documents that might contribute to model behavior. Our results supplement past observational analyses that link cooccurrence to model behavior, while demonstrating that extant methods for identifying relevant training documents do not fully explain an LM’s abilities and biases. Researchers can follow the recipe to test further hypotheses about how training data affects model behavior. Our code is made publicly available to promote future work.1</abstract>
<identifier type="citekey">nadkarni-etal-2026-rewriting</identifier>
<identifier type="doi">10.1162/tacl.a.740</identifier>
<location>
<url>https://aclanthology.org/2026.tacl-1.70/</url>
</location>
<part>
<date>2026</date>
<detail type="volume"><number>14</number></detail>
<extent unit="page">
<start>1562</start>
<end>1588</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T Rewriting History: A Recipe for Interventional Analyses to Study Data Effects on Model Behavior
%A Nadkarni, Rahul
%A Elazar, Yanai
%A Gonen, Hila
%A Smith, Noah A.
%J Transactions of the Association for Computational Linguistics
%D 2026
%V 14
%I MIT Press
%C Cambridge, MA
%F nadkarni-etal-2026-rewriting
%X We present an experimental recipe for studying the relationship between training data and language model (LM) behavior. We outline steps for intervening on data batches – i.e., “rewriting history” – and then retraining model checkpoints over that data to test hypotheses relating data to behavior. Our intervention recipe’s stages are (1) selecting evaluation items from a benchmark that measures model behavior, (2) matching relevant documents to those items, and (3) modifying those documents before retraining and measuring the effects. We demonstrate the utility of our recipe through case studies on factual knowledge acquisition and gender bias in LMs, using both cooccurrence statistics and information retrieval methods to identify documents that might contribute to model behavior. Our results supplement past observational analyses that link cooccurrence to model behavior, while demonstrating that extant methods for identifying relevant training documents do not fully explain an LM’s abilities and biases. Researchers can follow the recipe to test further hypotheses about how training data affects model behavior. Our code is made publicly available to promote future work.1
%R 10.1162/tacl.a.740
%U https://aclanthology.org/2026.tacl-1.70/
%U https://doi.org/10.1162/tacl.a.740
%P 1562-1588
Markdown (Informal)
[Rewriting History: A Recipe for Interventional Analyses to Study Data Effects on Model Behavior](https://aclanthology.org/2026.tacl-1.70/) (Nadkarni et al., TACL 2026)
ACL