@article{ding-etal-2026-friend,
title = "Friend or Foe: How {LLM}s' Safety Mind Gets Fooled by Intent Shift Attack",
author = "Ding, Peng and
Kuang, Jun and
Sun, Wen and
Wang, Zongyu and
Cao, Xuezhi and
Cai, Xunliang and
Chen, Jiajun and
Huang, Shujian",
journal = "Transactions of the Association for Computational Linguistics",
volume = "14",
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.tacl-1.108/",
doi = "10.1162/tacl.a.805",
pages = "2357--2373",
abstract = "Large language models (LLMs) remain vulnerable to jailbreaking attacks despite their impressive capabilities. Investigating these vulnerabilities is crucial for developing robust safety mechanisms. Existing attacks primarily bypass LLM safeguards by introducing additional context or adversarial tokens, while leaving the core harmful intent intact. In this paper, we introduce ISA (Intent Shift Attack), which conceals malicious intent from LLMs through subtle edits. More specifically, we establish a taxonomy of intent transformations and leverage them to generate attacks that may be misperceived by LLMs as benign requests. Unlike prior methods relying on complex tokens or lengthy context, our approach requires only limited modifications to the original request, yielding natural, fluent, and seemingly harmless prompts. Extensive experiments on both open-source and commercial LLMs demonstrate that ISA effectively induces safety failures, achieving high attack success rates through simple linguistic transformations. More critically, fine-tuning models solely on benign data reformulated with ISA templates elevates the success rates to nearly 100{\%}. For defense, we evaluate existing methods and demonstrate their limited effectiveness against ISA, while exploring both training-free and training-based mitigation strategies. Our findings reveal fundamental challenges in intent inference for LLM safety and underscore the need for more effective defenses. Our code and datasets are available at https://github.com/NJUNLP/ISA."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="ding-etal-2026-friend">
<titleInfo>
<title>Friend or Foe: How LLMs’ Safety Mind Gets Fooled by Intent Shift Attack</title>
</titleInfo>
<name type="personal">
<namePart type="given">Peng</namePart>
<namePart type="family">Ding</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jun</namePart>
<namePart type="family">Kuang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Wen</namePart>
<namePart type="family">Sun</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Zongyu</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Xuezhi</namePart>
<namePart type="family">Cao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Xunliang</namePart>
<namePart type="family">Cai</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jiajun</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shujian</namePart>
<namePart type="family">Huang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>Large language models (LLMs) remain vulnerable to jailbreaking attacks despite their impressive capabilities. Investigating these vulnerabilities is crucial for developing robust safety mechanisms. Existing attacks primarily bypass LLM safeguards by introducing additional context or adversarial tokens, while leaving the core harmful intent intact. In this paper, we introduce ISA (Intent Shift Attack), which conceals malicious intent from LLMs through subtle edits. More specifically, we establish a taxonomy of intent transformations and leverage them to generate attacks that may be misperceived by LLMs as benign requests. Unlike prior methods relying on complex tokens or lengthy context, our approach requires only limited modifications to the original request, yielding natural, fluent, and seemingly harmless prompts. Extensive experiments on both open-source and commercial LLMs demonstrate that ISA effectively induces safety failures, achieving high attack success rates through simple linguistic transformations. More critically, fine-tuning models solely on benign data reformulated with ISA templates elevates the success rates to nearly 100%. For defense, we evaluate existing methods and demonstrate their limited effectiveness against ISA, while exploring both training-free and training-based mitigation strategies. Our findings reveal fundamental challenges in intent inference for LLM safety and underscore the need for more effective defenses. Our code and datasets are available at https://github.com/NJUNLP/ISA.</abstract>
<identifier type="citekey">ding-etal-2026-friend</identifier>
<identifier type="doi">10.1162/tacl.a.805</identifier>
<location>
<url>https://aclanthology.org/2026.tacl-1.108/</url>
</location>
<part>
<date>2026</date>
<detail type="volume"><number>14</number></detail>
<extent unit="page">
<start>2357</start>
<end>2373</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T Friend or Foe: How LLMs’ Safety Mind Gets Fooled by Intent Shift Attack
%A Ding, Peng
%A Kuang, Jun
%A Sun, Wen
%A Wang, Zongyu
%A Cao, Xuezhi
%A Cai, Xunliang
%A Chen, Jiajun
%A Huang, Shujian
%J Transactions of the Association for Computational Linguistics
%D 2026
%V 14
%I MIT Press
%C Cambridge, MA
%F ding-etal-2026-friend
%X Large language models (LLMs) remain vulnerable to jailbreaking attacks despite their impressive capabilities. Investigating these vulnerabilities is crucial for developing robust safety mechanisms. Existing attacks primarily bypass LLM safeguards by introducing additional context or adversarial tokens, while leaving the core harmful intent intact. In this paper, we introduce ISA (Intent Shift Attack), which conceals malicious intent from LLMs through subtle edits. More specifically, we establish a taxonomy of intent transformations and leverage them to generate attacks that may be misperceived by LLMs as benign requests. Unlike prior methods relying on complex tokens or lengthy context, our approach requires only limited modifications to the original request, yielding natural, fluent, and seemingly harmless prompts. Extensive experiments on both open-source and commercial LLMs demonstrate that ISA effectively induces safety failures, achieving high attack success rates through simple linguistic transformations. More critically, fine-tuning models solely on benign data reformulated with ISA templates elevates the success rates to nearly 100%. For defense, we evaluate existing methods and demonstrate their limited effectiveness against ISA, while exploring both training-free and training-based mitigation strategies. Our findings reveal fundamental challenges in intent inference for LLM safety and underscore the need for more effective defenses. Our code and datasets are available at https://github.com/NJUNLP/ISA.
%R 10.1162/tacl.a.805
%U https://aclanthology.org/2026.tacl-1.108/
%U https://doi.org/10.1162/tacl.a.805
%P 2357-2373
Markdown (Informal)
[Friend or Foe: How LLMs’ Safety Mind Gets Fooled by Intent Shift Attack](https://aclanthology.org/2026.tacl-1.108/) (Ding et al., TACL 2026)
ACL