@inproceedings{maia-costa-2026-bootstrapping,
title = "Bootstrapping Text Anomaly Detection with {LLM}-Generated Weak Supervision",
author = "Maia, Fabio Masaracchia and
Costa, Anna Helena Reali",
editor = "Barbosa, Bryan Khelven da Silva and
Paes, Aline and
Felippo, Ariani Di",
booktitle = "Proceedings of the 17th {B}razilian Symposium in Information and Human Language Technology",
month = oct,
year = "2026",
address = "Cuiab{\'a}, Mato Grosso, Brazil",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.stil-1.20/",
doi = "10.5753/stil.2026.26601",
pages = "231--244",
abstract = "Text anomaly detection is challenging because anomalous instances often share vocabulary and surface form with normal data, making them hard to distinguish without semantic understanding. Semi-supervised methods can significantly outperform unsupervised baselines, but rely on labeled anomalies that are rarely available in practice. LLMs encode rich semantic knowledge that can approximate human judgments, yet using them directly as detectors is costly at inference time and sensitive to prompt design, while generating synthetic outliers risks distribution mismatch with real anomalies. We propose a different strategy: treating a compact, locally deployed LLM as a noisy annotator over real data. The LLM scores a small subset of unlabeled documents once at training time, producing weak labels that refine a sentence encoder via contrastive fine-tuning and train a lightweight downstream detector {---} without cloud APIs, human annotation, or curated anomaly datasets. Across four datasets spanning two languages and two tasks, the approach recovers up to 79{\%} of the gap to oracle upper bounds while using 14{--}91{\texttimes} fewer labeled anomalies. We further identify three failure modes that explain when LLM-generated weak supervision succeeds or breaks down."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="maia-costa-2026-bootstrapping">
<titleInfo>
<title>Bootstrapping Text Anomaly Detection with LLM-Generated Weak Supervision</title>
</titleInfo>
<name type="personal">
<namePart type="given">Fabio</namePart>
<namePart type="given">Masaracchia</namePart>
<namePart type="family">Maia</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anna</namePart>
<namePart type="given">Helena</namePart>
<namePart type="given">Reali</namePart>
<namePart type="family">Costa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bryan</namePart>
<namePart type="given">Khelven</namePart>
<namePart type="given">da</namePart>
<namePart type="given">Silva</namePart>
<namePart type="family">Barbosa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ariani</namePart>
<namePart type="given">Di</namePart>
<namePart type="family">Felippo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Cuiabá, Mato Grosso, Brazil</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Text anomaly detection is challenging because anomalous instances often share vocabulary and surface form with normal data, making them hard to distinguish without semantic understanding. Semi-supervised methods can significantly outperform unsupervised baselines, but rely on labeled anomalies that are rarely available in practice. LLMs encode rich semantic knowledge that can approximate human judgments, yet using them directly as detectors is costly at inference time and sensitive to prompt design, while generating synthetic outliers risks distribution mismatch with real anomalies. We propose a different strategy: treating a compact, locally deployed LLM as a noisy annotator over real data. The LLM scores a small subset of unlabeled documents once at training time, producing weak labels that refine a sentence encoder via contrastive fine-tuning and train a lightweight downstream detector — without cloud APIs, human annotation, or curated anomaly datasets. Across four datasets spanning two languages and two tasks, the approach recovers up to 79% of the gap to oracle upper bounds while using 14–91× fewer labeled anomalies. We further identify three failure modes that explain when LLM-generated weak supervision succeeds or breaks down.</abstract>
<identifier type="citekey">maia-costa-2026-bootstrapping</identifier>
<identifier type="doi">10.5753/stil.2026.26601</identifier>
<location>
<url>https://aclanthology.org/2026.stil-1.20/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>231</start>
<end>244</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Bootstrapping Text Anomaly Detection with LLM-Generated Weak Supervision
%A Maia, Fabio Masaracchia
%A Costa, Anna Helena Reali
%Y Barbosa, Bryan Khelven da Silva
%Y Paes, Aline
%Y Felippo, Ariani Di
%S Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology
%D 2026
%8 October
%I Association for Computational Linguistics
%C Cuiabá, Mato Grosso, Brazil
%F maia-costa-2026-bootstrapping
%X Text anomaly detection is challenging because anomalous instances often share vocabulary and surface form with normal data, making them hard to distinguish without semantic understanding. Semi-supervised methods can significantly outperform unsupervised baselines, but rely on labeled anomalies that are rarely available in practice. LLMs encode rich semantic knowledge that can approximate human judgments, yet using them directly as detectors is costly at inference time and sensitive to prompt design, while generating synthetic outliers risks distribution mismatch with real anomalies. We propose a different strategy: treating a compact, locally deployed LLM as a noisy annotator over real data. The LLM scores a small subset of unlabeled documents once at training time, producing weak labels that refine a sentence encoder via contrastive fine-tuning and train a lightweight downstream detector — without cloud APIs, human annotation, or curated anomaly datasets. Across four datasets spanning two languages and two tasks, the approach recovers up to 79% of the gap to oracle upper bounds while using 14–91× fewer labeled anomalies. We further identify three failure modes that explain when LLM-generated weak supervision succeeds or breaks down.
%R 10.5753/stil.2026.26601
%U https://aclanthology.org/2026.stil-1.20/
%U https://doi.org/10.5753/stil.2026.26601
%P 231-244
Markdown (Informal)
[Bootstrapping Text Anomaly Detection with LLM-Generated Weak Supervision](https://aclanthology.org/2026.stil-1.20/) (Maia & Costa, STIL 2026)
ACL