@article{ochs-habernal-2026-conundrum,
title = "The Conundrum of Trustworthy Research on Attacking Personally Identifiable Information Removal Techniques",
author = "Ochs, Sebastian and
Habernal, Ivan",
journal = "Computational Linguistics",
volume = "52",
number = "2",
month = jun,
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.cl-2.8/",
doi = "10.1162/coli.a.615",
pages = "725--757",
abstract = "Removing personally identifiable information (PII) from texts is necessary to comply with various data protection regulations and to enable data sharing without compromising privacy. However, recent works show that documents sanitized by PII-removal techniques are vulnerable to reconstruction attacks. Yet, we suspect that the reported success of these attacks is largely overestimated. We critically analyze the evaluation of existing attacks and find that data leakage and data contamination are not properly mitigated, leaving the question whether or not PII removal techniques truly protect privacy in real-world scenarios unaddressed. We investigate possible data sources and attack setups that avoid data leakage and conclude that only truly private data can allow us to objectively evaluate vulnerabilities in PII removal techniques. However, access to private data is heavily restricted{---}and for good reasons{---}which also means that the public research community cannot address this problem in a transparent, reproducible, and trustworthy manner."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="ochs-habernal-2026-conundrum">
<titleInfo>
<title>The Conundrum of Trustworthy Research on Attacking Personally Identifiable Information Removal Techniques</title>
</titleInfo>
<name type="personal">
<namePart type="given">Sebastian</namePart>
<namePart type="family">Ochs</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ivan</namePart>
<namePart type="family">Habernal</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>Removing personally identifiable information (PII) from texts is necessary to comply with various data protection regulations and to enable data sharing without compromising privacy. However, recent works show that documents sanitized by PII-removal techniques are vulnerable to reconstruction attacks. Yet, we suspect that the reported success of these attacks is largely overestimated. We critically analyze the evaluation of existing attacks and find that data leakage and data contamination are not properly mitigated, leaving the question whether or not PII removal techniques truly protect privacy in real-world scenarios unaddressed. We investigate possible data sources and attack setups that avoid data leakage and conclude that only truly private data can allow us to objectively evaluate vulnerabilities in PII removal techniques. However, access to private data is heavily restricted—and for good reasons—which also means that the public research community cannot address this problem in a transparent, reproducible, and trustworthy manner.</abstract>
<identifier type="citekey">ochs-habernal-2026-conundrum</identifier>
<identifier type="doi">10.1162/coli.a.615</identifier>
<location>
<url>https://aclanthology.org/2026.cl-2.8/</url>
</location>
<part>
<date>2026-06</date>
<detail type="volume"><number>52</number></detail>
<detail type="issue"><number>2</number></detail>
<extent unit="page">
<start>725</start>
<end>757</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T The Conundrum of Trustworthy Research on Attacking Personally Identifiable Information Removal Techniques
%A Ochs, Sebastian
%A Habernal, Ivan
%J Computational Linguistics
%D 2026
%8 June
%V 52
%N 2
%I MIT Press
%C Cambridge, MA
%F ochs-habernal-2026-conundrum
%X Removing personally identifiable information (PII) from texts is necessary to comply with various data protection regulations and to enable data sharing without compromising privacy. However, recent works show that documents sanitized by PII-removal techniques are vulnerable to reconstruction attacks. Yet, we suspect that the reported success of these attacks is largely overestimated. We critically analyze the evaluation of existing attacks and find that data leakage and data contamination are not properly mitigated, leaving the question whether or not PII removal techniques truly protect privacy in real-world scenarios unaddressed. We investigate possible data sources and attack setups that avoid data leakage and conclude that only truly private data can allow us to objectively evaluate vulnerabilities in PII removal techniques. However, access to private data is heavily restricted—and for good reasons—which also means that the public research community cannot address this problem in a transparent, reproducible, and trustworthy manner.
%R 10.1162/coli.a.615
%U https://aclanthology.org/2026.cl-2.8/
%U https://doi.org/10.1162/coli.a.615
%P 725-757
Markdown (Informal)
[The Conundrum of Trustworthy Research on Attacking Personally Identifiable Information Removal Techniques](https://aclanthology.org/2026.cl-2.8/) (Ochs & Habernal, CL 2026)
ACL