@inproceedings{cherif-etal-2026-llm,
title = "{LLM}-based Defense Against Adversarial Abstracts in {ML}/{AI} Conference Reviewer Assignments",
author = "Cherif, Mohamed Omar and
Cerisara, Christophe and
Falgas, Julien",
editor = "Mitkov, Ruslan and
Mu{\~n}oz, Rafael and
Lloret, Elena and
Ranasinghe, Tharindu and
Estevanell-Valladares, Ernesto L. and
Lamsiyah, Salima and
Montoyo, Andr{\'e}s and
Ezzini, Saad",
booktitle = "Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security",
month = jun,
year = "2026",
address = "Alicante, Spain",
publisher = "Department of Languages and Information Systems, University of Alicante",
url = "https://aclanthology.org/2026.nlpaics-1.12/",
pages = "113--121",
abstract = "Large Machine Learning Conferences guarantee quality reviews of submitted papers by assigning each submission to reviewers who are experts in the relevant topics. This is typically realized by matching reviewers' expertise to the paper abstract with text semantic embeddings. In order to try and maximize their acceptance score, malevolent authors may attack this assignment process by modifying their abstract so that it matches the expertise of colluding accomplices registered as reviewers. The success of such attacks has been recently demonstrated for realistic conference reviewing datasets with SPECTER embeddings. We propose in this work a defense mechanism against such attacks that leverages Large Language Models (LLM) to rewrite the submitted abstracts and remove the targeted alteration of the original abstract that were aimed at the colluding reviewers. We demonstrate experimentally the effectiveness of our defense that prevents assigning most malevolent abstracts to their colluding reviewer, while preserving the topic-based assignment of normal abstracts to expert reviewers."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="cherif-etal-2026-llm">
<titleInfo>
<title>LLM-based Defense Against Adversarial Abstracts in ML/AI Conference Reviewer Assignments</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mohamed</namePart>
<namePart type="given">Omar</namePart>
<namePart type="family">Cherif</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christophe</namePart>
<namePart type="family">Cerisara</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Julien</namePart>
<namePart type="family">Falgas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ruslan</namePart>
<namePart type="family">Mitkov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rafael</namePart>
<namePart type="family">Muñoz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Lloret</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tharindu</namePart>
<namePart type="family">Ranasinghe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ernesto</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Estevanell-Valladares</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Salima</namePart>
<namePart type="family">Lamsiyah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andrés</namePart>
<namePart type="family">Montoyo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Department of Languages and Information Systems, University of Alicante</publisher>
<place>
<placeTerm type="text">Alicante, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Large Machine Learning Conferences guarantee quality reviews of submitted papers by assigning each submission to reviewers who are experts in the relevant topics. This is typically realized by matching reviewers’ expertise to the paper abstract with text semantic embeddings. In order to try and maximize their acceptance score, malevolent authors may attack this assignment process by modifying their abstract so that it matches the expertise of colluding accomplices registered as reviewers. The success of such attacks has been recently demonstrated for realistic conference reviewing datasets with SPECTER embeddings. We propose in this work a defense mechanism against such attacks that leverages Large Language Models (LLM) to rewrite the submitted abstracts and remove the targeted alteration of the original abstract that were aimed at the colluding reviewers. We demonstrate experimentally the effectiveness of our defense that prevents assigning most malevolent abstracts to their colluding reviewer, while preserving the topic-based assignment of normal abstracts to expert reviewers.</abstract>
<identifier type="citekey">cherif-etal-2026-llm</identifier>
<location>
<url>https://aclanthology.org/2026.nlpaics-1.12/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>113</start>
<end>121</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T LLM-based Defense Against Adversarial Abstracts in ML/AI Conference Reviewer Assignments
%A Cherif, Mohamed Omar
%A Cerisara, Christophe
%A Falgas, Julien
%Y Mitkov, Ruslan
%Y Muñoz, Rafael
%Y Lloret, Elena
%Y Ranasinghe, Tharindu
%Y Estevanell-Valladares, Ernesto L.
%Y Lamsiyah, Salima
%Y Montoyo, Andrés
%Y Ezzini, Saad
%S Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security
%D 2026
%8 June
%I Department of Languages and Information Systems, University of Alicante
%C Alicante, Spain
%F cherif-etal-2026-llm
%X Large Machine Learning Conferences guarantee quality reviews of submitted papers by assigning each submission to reviewers who are experts in the relevant topics. This is typically realized by matching reviewers’ expertise to the paper abstract with text semantic embeddings. In order to try and maximize their acceptance score, malevolent authors may attack this assignment process by modifying their abstract so that it matches the expertise of colluding accomplices registered as reviewers. The success of such attacks has been recently demonstrated for realistic conference reviewing datasets with SPECTER embeddings. We propose in this work a defense mechanism against such attacks that leverages Large Language Models (LLM) to rewrite the submitted abstracts and remove the targeted alteration of the original abstract that were aimed at the colluding reviewers. We demonstrate experimentally the effectiveness of our defense that prevents assigning most malevolent abstracts to their colluding reviewer, while preserving the topic-based assignment of normal abstracts to expert reviewers.
%U https://aclanthology.org/2026.nlpaics-1.12/
%P 113-121
Markdown (Informal)
[LLM-based Defense Against Adversarial Abstracts in ML/AI Conference Reviewer Assignments](https://aclanthology.org/2026.nlpaics-1.12/) (Cherif et al., NLPAICS 2026)
ACL