@inproceedings{stepankova-etal-2026-semantic,
title = "Semantic Clustering of Obfuscated Command-Line Detections for Alert Reduction",
author = "{\v{S}}t{\v{e}}p{\'a}nkov{\'a}, Barbora and
Outrata, Vojt{\v{e}}ch and
Kopp, Martin",
editor = "Mitkov, Ruslan and
Mu{\~n}oz, Rafael and
Lloret, Elena and
Ranasinghe, Tharindu and
Estevanell-Valladares, Ernesto L. and
Lamsiyah, Salima and
Montoyo, Andr{\'e}s and
Ezzini, Saad",
booktitle = "Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security",
month = jun,
year = "2026",
address = "Alicante, Spain",
publisher = "Department of Languages and Information Systems, University of Alicante",
url = "https://aclanthology.org/2026.nlpaics-1.19/",
pages = "176--183",
abstract = "We propose a post-processing method for grouping large volumes of command-line detections into semantically coherent cluster-level alerts. The approach combines embedding-based clustering with LLM-based cluster-level filtering: command-lines are first encoded using Sentence-BERT embeddings and grouped via hierarchical agglomerative clustering, after which a large language model evaluates cluster representatives to reduce the number of false positive alerts. We evaluate the method on real-world telemetry from a commercial endpoint protection system, applying it to detections produced by an obfuscation detection model. On one week of data, the pipeline reduces alert volume by approximately 98{\%} while maintaining high cluster purity and semantic coherence, demonstrating its effectiveness as a scalable post-processing step in high-volume detection settings."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="stepankova-etal-2026-semantic">
<titleInfo>
<title>Semantic Clustering of Obfuscated Command-Line Detections for Alert Reduction</title>
</titleInfo>
<name type="personal">
<namePart type="given">Barbora</namePart>
<namePart type="family">Štěpánková</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vojtěch</namePart>
<namePart type="family">Outrata</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Martin</namePart>
<namePart type="family">Kopp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ruslan</namePart>
<namePart type="family">Mitkov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rafael</namePart>
<namePart type="family">Muñoz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Lloret</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tharindu</namePart>
<namePart type="family">Ranasinghe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ernesto</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Estevanell-Valladares</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Salima</namePart>
<namePart type="family">Lamsiyah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andrés</namePart>
<namePart type="family">Montoyo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Department of Languages and Information Systems, University of Alicante</publisher>
<place>
<placeTerm type="text">Alicante, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We propose a post-processing method for grouping large volumes of command-line detections into semantically coherent cluster-level alerts. The approach combines embedding-based clustering with LLM-based cluster-level filtering: command-lines are first encoded using Sentence-BERT embeddings and grouped via hierarchical agglomerative clustering, after which a large language model evaluates cluster representatives to reduce the number of false positive alerts. We evaluate the method on real-world telemetry from a commercial endpoint protection system, applying it to detections produced by an obfuscation detection model. On one week of data, the pipeline reduces alert volume by approximately 98% while maintaining high cluster purity and semantic coherence, demonstrating its effectiveness as a scalable post-processing step in high-volume detection settings.</abstract>
<identifier type="citekey">stepankova-etal-2026-semantic</identifier>
<location>
<url>https://aclanthology.org/2026.nlpaics-1.19/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>176</start>
<end>183</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Semantic Clustering of Obfuscated Command-Line Detections for Alert Reduction
%A Štěpánková, Barbora
%A Outrata, Vojtěch
%A Kopp, Martin
%Y Mitkov, Ruslan
%Y Muñoz, Rafael
%Y Lloret, Elena
%Y Ranasinghe, Tharindu
%Y Estevanell-Valladares, Ernesto L.
%Y Lamsiyah, Salima
%Y Montoyo, Andrés
%Y Ezzini, Saad
%S Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security
%D 2026
%8 June
%I Department of Languages and Information Systems, University of Alicante
%C Alicante, Spain
%F stepankova-etal-2026-semantic
%X We propose a post-processing method for grouping large volumes of command-line detections into semantically coherent cluster-level alerts. The approach combines embedding-based clustering with LLM-based cluster-level filtering: command-lines are first encoded using Sentence-BERT embeddings and grouped via hierarchical agglomerative clustering, after which a large language model evaluates cluster representatives to reduce the number of false positive alerts. We evaluate the method on real-world telemetry from a commercial endpoint protection system, applying it to detections produced by an obfuscation detection model. On one week of data, the pipeline reduces alert volume by approximately 98% while maintaining high cluster purity and semantic coherence, demonstrating its effectiveness as a scalable post-processing step in high-volume detection settings.
%U https://aclanthology.org/2026.nlpaics-1.19/
%P 176-183
Markdown (Informal)
[Semantic Clustering of Obfuscated Command-Line Detections for Alert Reduction](https://aclanthology.org/2026.nlpaics-1.19/) (Štěpánková et al., NLPAICS 2026)
ACL