@article{saha-etal-2026-breaking,
title = "Breaking the Code: Security Assessment of {AI} Code Agents Through Systematic Jailbreaking Attacks",
author = "Saha, Shoumik and
Chen, Jifan and
Mayers, Sam and
Gouda, Sanjay Krishna and
Wang, Zijian and
Kumar, Varun",
journal = "Transactions of the Association for Computational Linguistics",
volume = "14",
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.tacl-1.94/",
doi = "10.1162/tacl.a.792",
pages = "2081--2102",
abstract = "Code-capable large language model (LLM) agents are embedded in software engineering workflows where they can read, write, and execute code, raising ``jailbreak'' stakes beyond text-only settings. Prior evaluations emphasize refusal or harmful-text detection, leaving open whether agents compile and run malicious programs. We present JAWS-BENCH(Jailbreaks Across WorkSpaces), a benchmark spanning three escalating workspace regimes mirroring attacker capability: empty (JAWS-0), single-file (JAWS-1), and multi-file (JAWS-M). We pair it with a hierarchical, executable-aware Judge Framework that tests (i) compliance, (ii) attack success, (iii) syntactic correctness, and (iv) runtime executability to measure de-ployable harm. Across seven LLM backends from five families, prompt-only attacks in JAWS-0 achieve 61{\%} compliance; 58{\%} are harmful, 52{\%} parse, and 27{\%} run end-to-end. In JAWS-1, compliance reaches {~} 100{\%} for stronger models with a mean ASR (Attack Success Rate) {\ensuremath{\approx}} 71{\%}; JAWS-M raises mean ASR to {\ensuremath{\approx}} 75{\%}, with 32{\%} runnable attack code. Wrapping an LLM in an agent increases ASR by 1.6{\texttimes}, by overturning initial refusals during planning and tool use. Additional evaluations with SWE-Agent and OpenAI Codex exhibit similar trends, indicating that JAWS-BENCH can be reused across multiple agent frameworks. Category analyses identify which attack classes are most vulnerable and deployable, motivating execution-aware defenses and refusal-preserving agent designs."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="saha-etal-2026-breaking">
<titleInfo>
<title>Breaking the Code: Security Assessment of AI Code Agents Through Systematic Jailbreaking Attacks</title>
</titleInfo>
<name type="personal">
<namePart type="given">Shoumik</namePart>
<namePart type="family">Saha</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jifan</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sam</namePart>
<namePart type="family">Mayers</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sanjay</namePart>
<namePart type="given">Krishna</namePart>
<namePart type="family">Gouda</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Zijian</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Varun</namePart>
<namePart type="family">Kumar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>Code-capable large language model (LLM) agents are embedded in software engineering workflows where they can read, write, and execute code, raising “jailbreak” stakes beyond text-only settings. Prior evaluations emphasize refusal or harmful-text detection, leaving open whether agents compile and run malicious programs. We present JAWS-BENCH(Jailbreaks Across WorkSpaces), a benchmark spanning three escalating workspace regimes mirroring attacker capability: empty (JAWS-0), single-file (JAWS-1), and multi-file (JAWS-M). We pair it with a hierarchical, executable-aware Judge Framework that tests (i) compliance, (ii) attack success, (iii) syntactic correctness, and (iv) runtime executability to measure de-ployable harm. Across seven LLM backends from five families, prompt-only attacks in JAWS-0 achieve 61% compliance; 58% are harmful, 52% parse, and 27% run end-to-end. In JAWS-1, compliance reaches 100% for stronger models with a mean ASR (Attack Success Rate) \ensuremath\approx 71%; JAWS-M raises mean ASR to \ensuremath\approx 75%, with 32% runnable attack code. Wrapping an LLM in an agent increases ASR by 1.6×, by overturning initial refusals during planning and tool use. Additional evaluations with SWE-Agent and OpenAI Codex exhibit similar trends, indicating that JAWS-BENCH can be reused across multiple agent frameworks. Category analyses identify which attack classes are most vulnerable and deployable, motivating execution-aware defenses and refusal-preserving agent designs.</abstract>
<identifier type="citekey">saha-etal-2026-breaking</identifier>
<identifier type="doi">10.1162/tacl.a.792</identifier>
<location>
<url>https://aclanthology.org/2026.tacl-1.94/</url>
</location>
<part>
<date>2026</date>
<detail type="volume"><number>14</number></detail>
<extent unit="page">
<start>2081</start>
<end>2102</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T Breaking the Code: Security Assessment of AI Code Agents Through Systematic Jailbreaking Attacks
%A Saha, Shoumik
%A Chen, Jifan
%A Mayers, Sam
%A Gouda, Sanjay Krishna
%A Wang, Zijian
%A Kumar, Varun
%J Transactions of the Association for Computational Linguistics
%D 2026
%V 14
%I MIT Press
%C Cambridge, MA
%F saha-etal-2026-breaking
%X Code-capable large language model (LLM) agents are embedded in software engineering workflows where they can read, write, and execute code, raising “jailbreak” stakes beyond text-only settings. Prior evaluations emphasize refusal or harmful-text detection, leaving open whether agents compile and run malicious programs. We present JAWS-BENCH(Jailbreaks Across WorkSpaces), a benchmark spanning three escalating workspace regimes mirroring attacker capability: empty (JAWS-0), single-file (JAWS-1), and multi-file (JAWS-M). We pair it with a hierarchical, executable-aware Judge Framework that tests (i) compliance, (ii) attack success, (iii) syntactic correctness, and (iv) runtime executability to measure de-ployable harm. Across seven LLM backends from five families, prompt-only attacks in JAWS-0 achieve 61% compliance; 58% are harmful, 52% parse, and 27% run end-to-end. In JAWS-1, compliance reaches 100% for stronger models with a mean ASR (Attack Success Rate) \ensuremath\approx 71%; JAWS-M raises mean ASR to \ensuremath\approx 75%, with 32% runnable attack code. Wrapping an LLM in an agent increases ASR by 1.6×, by overturning initial refusals during planning and tool use. Additional evaluations with SWE-Agent and OpenAI Codex exhibit similar trends, indicating that JAWS-BENCH can be reused across multiple agent frameworks. Category analyses identify which attack classes are most vulnerable and deployable, motivating execution-aware defenses and refusal-preserving agent designs.
%R 10.1162/tacl.a.792
%U https://aclanthology.org/2026.tacl-1.94/
%U https://doi.org/10.1162/tacl.a.792
%P 2081-2102
Markdown (Informal)
[Breaking the Code: Security Assessment of AI Code Agents Through Systematic Jailbreaking Attacks](https://aclanthology.org/2026.tacl-1.94/) (Saha et al., TACL 2026)
ACL