@inproceedings{deng-etal-2026-flash,
title = "{FLASH}: Focused Layer Attention Sink Hijacking",
author = "Deng, Siyuan and
Li, Yerong and
Shang, Fanhua and
Liu, Hongying",
editor = "Liakata, Maria and
Moreira, Viviane P. and
Zhang, Jiajun and
Jurgens, David",
booktitle = "Findings of the {A}ssociation for {C}omputational {L}inguistics: {ACL} 2026",
month = jul,
year = "2026",
address = "San Diego, California, United States",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.findings-acl.1039/",
doi = "10.18653/v1/2026.findings-acl.1039",
pages = "20743--20761",
ISBN = "979-8-89176-395-1",
abstract = "Despite advances in safety alignment, Large Language Models (LLMs) remain vulnerable to jailbreaking attacks. However, prevailing methods suffer from a dichotomy of limitations: they either rely on prohibitive iterative optimization in the input space (leading to high computational costs) or fail to penetrate the model{'}s internal decision-making processes. In this work, we identify a critical structural vulnerability: the ``Attention Sink'' mechanism{---}originally designed to maintain generation stability by anchoring to initial tokens{---}unintentionally serves as a computational anchor for safety alignment. We hypothesize and empirically verify that safety guardrails are not globally distributed but are predominantly ``front-loaded'' in specific attention heads of shallow layers. To audit this structural fragility, we propose FLASH (Focused Layer Attention Sink Hijacking), a novel diagnostic auditing framework that executes a surgical intervention. By precisely scaling attention scores in these vulnerable layers, we dismantle the model{'}s internal safety anchor. To ensure the attack{'}s robustness and coherence against the resulting internal noise, we synergize this intervention with multi-variant query rewriting and an adaptive dynamic decoding strategy. Extensive experiments on Llama-3, Qwen-3, and others demonstrate that FLASH achieves a state-of-the-art Attack Success Rate of over 77{\%} with an unprecedented efficiency of 1.53 queries on average. This work marks a paradigm shift from brute-force optimization to mechanism-driven diagnostic auditing, exposing a fundamental trade-off between architectural stability and safety security."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="deng-etal-2026-flash">
<titleInfo>
<title>FLASH: Focused Layer Attention Sink Hijacking</title>
</titleInfo>
<name type="personal">
<namePart type="given">Siyuan</namePart>
<namePart type="family">Deng</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yerong</namePart>
<namePart type="family">Li</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fanhua</namePart>
<namePart type="family">Shang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hongying</namePart>
<namePart type="family">Liu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-07</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Findings of the Association for Computational Linguistics: ACL 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="family">Liakata</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Viviane</namePart>
<namePart type="given">P</namePart>
<namePart type="family">Moreira</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jiajun</namePart>
<namePart type="family">Zhang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">David</namePart>
<namePart type="family">Jurgens</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">San Diego, California, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-89176-395-1</identifier>
</relatedItem>
<abstract>Despite advances in safety alignment, Large Language Models (LLMs) remain vulnerable to jailbreaking attacks. However, prevailing methods suffer from a dichotomy of limitations: they either rely on prohibitive iterative optimization in the input space (leading to high computational costs) or fail to penetrate the model’s internal decision-making processes. In this work, we identify a critical structural vulnerability: the “Attention Sink” mechanism—originally designed to maintain generation stability by anchoring to initial tokens—unintentionally serves as a computational anchor for safety alignment. We hypothesize and empirically verify that safety guardrails are not globally distributed but are predominantly “front-loaded” in specific attention heads of shallow layers. To audit this structural fragility, we propose FLASH (Focused Layer Attention Sink Hijacking), a novel diagnostic auditing framework that executes a surgical intervention. By precisely scaling attention scores in these vulnerable layers, we dismantle the model’s internal safety anchor. To ensure the attack’s robustness and coherence against the resulting internal noise, we synergize this intervention with multi-variant query rewriting and an adaptive dynamic decoding strategy. Extensive experiments on Llama-3, Qwen-3, and others demonstrate that FLASH achieves a state-of-the-art Attack Success Rate of over 77% with an unprecedented efficiency of 1.53 queries on average. This work marks a paradigm shift from brute-force optimization to mechanism-driven diagnostic auditing, exposing a fundamental trade-off between architectural stability and safety security.</abstract>
<identifier type="citekey">deng-etal-2026-flash</identifier>
<identifier type="doi">10.18653/v1/2026.findings-acl.1039</identifier>
<location>
<url>https://aclanthology.org/2026.findings-acl.1039/</url>
</location>
<part>
<date>2026-07</date>
<extent unit="page">
<start>20743</start>
<end>20761</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T FLASH: Focused Layer Attention Sink Hijacking
%A Deng, Siyuan
%A Li, Yerong
%A Shang, Fanhua
%A Liu, Hongying
%Y Liakata, Maria
%Y Moreira, Viviane P.
%Y Zhang, Jiajun
%Y Jurgens, David
%S Findings of the Association for Computational Linguistics: ACL 2026
%D 2026
%8 July
%I Association for Computational Linguistics
%C San Diego, California, United States
%@ 979-8-89176-395-1
%F deng-etal-2026-flash
%X Despite advances in safety alignment, Large Language Models (LLMs) remain vulnerable to jailbreaking attacks. However, prevailing methods suffer from a dichotomy of limitations: they either rely on prohibitive iterative optimization in the input space (leading to high computational costs) or fail to penetrate the model’s internal decision-making processes. In this work, we identify a critical structural vulnerability: the “Attention Sink” mechanism—originally designed to maintain generation stability by anchoring to initial tokens—unintentionally serves as a computational anchor for safety alignment. We hypothesize and empirically verify that safety guardrails are not globally distributed but are predominantly “front-loaded” in specific attention heads of shallow layers. To audit this structural fragility, we propose FLASH (Focused Layer Attention Sink Hijacking), a novel diagnostic auditing framework that executes a surgical intervention. By precisely scaling attention scores in these vulnerable layers, we dismantle the model’s internal safety anchor. To ensure the attack’s robustness and coherence against the resulting internal noise, we synergize this intervention with multi-variant query rewriting and an adaptive dynamic decoding strategy. Extensive experiments on Llama-3, Qwen-3, and others demonstrate that FLASH achieves a state-of-the-art Attack Success Rate of over 77% with an unprecedented efficiency of 1.53 queries on average. This work marks a paradigm shift from brute-force optimization to mechanism-driven diagnostic auditing, exposing a fundamental trade-off between architectural stability and safety security.
%R 10.18653/v1/2026.findings-acl.1039
%U https://aclanthology.org/2026.findings-acl.1039/
%U https://doi.org/10.18653/v1/2026.findings-acl.1039
%P 20743-20761
Markdown (Informal)
[FLASH: Focused Layer Attention Sink Hijacking](https://aclanthology.org/2026.findings-acl.1039/) (Deng et al., Findings 2026)
ACL
- Siyuan Deng, Yerong Li, Fanhua Shang, and Hongying Liu. 2026. FLASH: Focused Layer Attention Sink Hijacking. In Findings of the Association for Computational Linguistics: ACL 2026, pages 20743–20761, San Diego, California, United States. Association for Computational Linguistics.