@inproceedings{wu-etal-2025-longattn,
title = "{L}ong{A}ttn: Selecting Long-context Training Data via Token-level Attention",
author = "Wu, Longyun and
Zhu, Dawei and
Zhao, Guangxiang and
Yu, Zhuocheng and
Ran, Junfeng and
Wong, Xiangyu and
Sun, Lin and
Li, Sujian",
editor = "Che, Wanxiang and
Nabende, Joyce and
Shutova, Ekaterina and
Pilehvar, Mohammad Taher",
booktitle = "Findings of the Association for Computational Linguistics: ACL 2025",
month = jul,
year = "2025",
address = "Vienna, Austria",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2025.findings-acl.991/",
doi = "10.18653/v1/2025.findings-acl.991",
pages = "19367--19380",
ISBN = "979-8-89176-256-5",
abstract = "With the development of large language models (LLMs), there has been an increasing need for significant advancements in handling long contexts. To enhance long-context capabilities, constructing high-quality training data with \textbf{long-range dependencies} is crucial. Existing methods to select long-context data often rely on sentence-level analysis,which can be greatly optimized in both performance and efficiency. In this paper, we propose a novel token-level framework, \textbf{LongAttn}, which leverages the self-attention mechanism of LLMs to measure the \textbf{long-range dependencies} for the data. By calculating token-level dependency strength and distribution uniformity of token scores, LongAttn effectively quantifies \textbf{long-range dependencies}, enabling more accurate and efficient data selection. We filter \textbf{LongABC-32K} from open-source long-context datasets (ArXiv, Book, and Code). Through our comprehensive experiments, LongAttn has demonstrated its excellent \textbf{effectiveness}, \textbf{scalability}, and \textbf{efficiency}. We will release our code and the high-quality long-context dataset \textbf{LongABC-32K} in the future."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="wu-etal-2025-longattn">
<titleInfo>
<title>LongAttn: Selecting Long-context Training Data via Token-level Attention</title>
</titleInfo>
<name type="personal">
<namePart type="given">Longyun</namePart>
<namePart type="family">Wu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dawei</namePart>
<namePart type="family">Zhu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Guangxiang</namePart>
<namePart type="family">Zhao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Zhuocheng</namePart>
<namePart type="family">Yu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Junfeng</namePart>
<namePart type="family">Ran</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Xiangyu</namePart>
<namePart type="family">Wong</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lin</namePart>
<namePart type="family">Sun</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sujian</namePart>
<namePart type="family">Li</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2025-07</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Findings of the Association for Computational Linguistics: ACL 2025</title>
</titleInfo>
<name type="personal">
<namePart type="given">Wanxiang</namePart>
<namePart type="family">Che</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joyce</namePart>
<namePart type="family">Nabende</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ekaterina</namePart>
<namePart type="family">Shutova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mohammad</namePart>
<namePart type="given">Taher</namePart>
<namePart type="family">Pilehvar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Vienna, Austria</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-89176-256-5</identifier>
</relatedItem>
<abstract>With the development of large language models (LLMs), there has been an increasing need for significant advancements in handling long contexts. To enhance long-context capabilities, constructing high-quality training data with long-range dependencies is crucial. Existing methods to select long-context data often rely on sentence-level analysis,which can be greatly optimized in both performance and efficiency. In this paper, we propose a novel token-level framework, LongAttn, which leverages the self-attention mechanism of LLMs to measure the long-range dependencies for the data. By calculating token-level dependency strength and distribution uniformity of token scores, LongAttn effectively quantifies long-range dependencies, enabling more accurate and efficient data selection. We filter LongABC-32K from open-source long-context datasets (ArXiv, Book, and Code). Through our comprehensive experiments, LongAttn has demonstrated its excellent effectiveness, scalability, and efficiency. We will release our code and the high-quality long-context dataset LongABC-32K in the future.</abstract>
<identifier type="citekey">wu-etal-2025-longattn</identifier>
<identifier type="doi">10.18653/v1/2025.findings-acl.991</identifier>
<location>
<url>https://aclanthology.org/2025.findings-acl.991/</url>
</location>
<part>
<date>2025-07</date>
<extent unit="page">
<start>19367</start>
<end>19380</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T LongAttn: Selecting Long-context Training Data via Token-level Attention
%A Wu, Longyun
%A Zhu, Dawei
%A Zhao, Guangxiang
%A Yu, Zhuocheng
%A Ran, Junfeng
%A Wong, Xiangyu
%A Sun, Lin
%A Li, Sujian
%Y Che, Wanxiang
%Y Nabende, Joyce
%Y Shutova, Ekaterina
%Y Pilehvar, Mohammad Taher
%S Findings of the Association for Computational Linguistics: ACL 2025
%D 2025
%8 July
%I Association for Computational Linguistics
%C Vienna, Austria
%@ 979-8-89176-256-5
%F wu-etal-2025-longattn
%X With the development of large language models (LLMs), there has been an increasing need for significant advancements in handling long contexts. To enhance long-context capabilities, constructing high-quality training data with long-range dependencies is crucial. Existing methods to select long-context data often rely on sentence-level analysis,which can be greatly optimized in both performance and efficiency. In this paper, we propose a novel token-level framework, LongAttn, which leverages the self-attention mechanism of LLMs to measure the long-range dependencies for the data. By calculating token-level dependency strength and distribution uniformity of token scores, LongAttn effectively quantifies long-range dependencies, enabling more accurate and efficient data selection. We filter LongABC-32K from open-source long-context datasets (ArXiv, Book, and Code). Through our comprehensive experiments, LongAttn has demonstrated its excellent effectiveness, scalability, and efficiency. We will release our code and the high-quality long-context dataset LongABC-32K in the future.
%R 10.18653/v1/2025.findings-acl.991
%U https://aclanthology.org/2025.findings-acl.991/
%U https://doi.org/10.18653/v1/2025.findings-acl.991
%P 19367-19380
Markdown (Informal)
[LongAttn: Selecting Long-context Training Data via Token-level Attention](https://aclanthology.org/2025.findings-acl.991/) (Wu et al., Findings 2025)
ACL
- Longyun Wu, Dawei Zhu, Guangxiang Zhao, Zhuocheng Yu, Junfeng Ran, Xiangyu Wong, Lin Sun, and Sujian Li. 2025. LongAttn: Selecting Long-context Training Data via Token-level Attention. In Findings of the Association for Computational Linguistics: ACL 2025, pages 19367–19380, Vienna, Austria. Association for Computational Linguistics.