@inproceedings{bingert-etal-2026-benchmarking,
title = "Benchmarking {LLM}s for {ARR} Area Assignment: Evidence and Implications for Assignment Strategies",
author = "Bingert, Eileen and
Alves, Diego and
Degaetano-Ortlieb, Stefania",
editor = "Rehm, Georg and
Dietze, Stefan and
Dessi, Danilo and
Maynard, Diana and
Schimmler, Sonja",
booktitle = "Proceedings of Natural Scientific Language Processing ({NSLP}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.nslp-1.2/",
doi = "10.63317/4jbm8o3us4c8",
pages = "13--24",
abstract = "We study how large language models (LLMs) perform at assigning ACL Rolling Review (ARR) areas from paper titles/abstracts. Using 558 papers (ACL/EACL/NAACL, 2020 to 2025), we compare multiple LLMs and prompting schemes (zero/few-shot; with/without ARR keywords; each-category variants) and analyze per-area scores, error overlap, and confusion matrices. One-shot prompting (with OpenAI-gpt-oss-20b) tends to perform best, while injecting ARR keywords often lowers accuracy. Task-bounded areas (e.g., MT, IE, QA, Summarization) are predicted more reliably, whereas broad, cross-cutting labels (e.g., Resources and Evaluation, NLP Applications) are frequently conflated, indicating taxonomy ambiguity rather than solely model limitations. We recommend hierarchical or primary-plus-secondary labels to reduce ambiguity and improve reviewer matching. Our dataset, methods, and findings offer a reproducible baseline for area selection support in ACL workflows."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="bingert-etal-2026-benchmarking">
<titleInfo>
<title>Benchmarking LLMs for ARR Area Assignment: Evidence and Implications for Assignment Strategies</title>
</titleInfo>
<name type="personal">
<namePart type="given">Eileen</namePart>
<namePart type="family">Bingert</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Diego</namePart>
<namePart type="family">Alves</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stefania</namePart>
<namePart type="family">Degaetano-Ortlieb</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Natural Scientific Language Processing (NSLP) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Georg</namePart>
<namePart type="family">Rehm</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stefan</namePart>
<namePart type="family">Dietze</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Danilo</namePart>
<namePart type="family">Dessi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Diana</namePart>
<namePart type="family">Maynard</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sonja</namePart>
<namePart type="family">Schimmler</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We study how large language models (LLMs) perform at assigning ACL Rolling Review (ARR) areas from paper titles/abstracts. Using 558 papers (ACL/EACL/NAACL, 2020 to 2025), we compare multiple LLMs and prompting schemes (zero/few-shot; with/without ARR keywords; each-category variants) and analyze per-area scores, error overlap, and confusion matrices. One-shot prompting (with OpenAI-gpt-oss-20b) tends to perform best, while injecting ARR keywords often lowers accuracy. Task-bounded areas (e.g., MT, IE, QA, Summarization) are predicted more reliably, whereas broad, cross-cutting labels (e.g., Resources and Evaluation, NLP Applications) are frequently conflated, indicating taxonomy ambiguity rather than solely model limitations. We recommend hierarchical or primary-plus-secondary labels to reduce ambiguity and improve reviewer matching. Our dataset, methods, and findings offer a reproducible baseline for area selection support in ACL workflows.</abstract>
<identifier type="citekey">bingert-etal-2026-benchmarking</identifier>
<identifier type="doi">10.63317/4jbm8o3us4c8</identifier>
<location>
<url>https://aclanthology.org/2026.nslp-1.2/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>13</start>
<end>24</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Benchmarking LLMs for ARR Area Assignment: Evidence and Implications for Assignment Strategies
%A Bingert, Eileen
%A Alves, Diego
%A Degaetano-Ortlieb, Stefania
%Y Rehm, Georg
%Y Dietze, Stefan
%Y Dessi, Danilo
%Y Maynard, Diana
%Y Schimmler, Sonja
%S Proceedings of Natural Scientific Language Processing (NSLP) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F bingert-etal-2026-benchmarking
%X We study how large language models (LLMs) perform at assigning ACL Rolling Review (ARR) areas from paper titles/abstracts. Using 558 papers (ACL/EACL/NAACL, 2020 to 2025), we compare multiple LLMs and prompting schemes (zero/few-shot; with/without ARR keywords; each-category variants) and analyze per-area scores, error overlap, and confusion matrices. One-shot prompting (with OpenAI-gpt-oss-20b) tends to perform best, while injecting ARR keywords often lowers accuracy. Task-bounded areas (e.g., MT, IE, QA, Summarization) are predicted more reliably, whereas broad, cross-cutting labels (e.g., Resources and Evaluation, NLP Applications) are frequently conflated, indicating taxonomy ambiguity rather than solely model limitations. We recommend hierarchical or primary-plus-secondary labels to reduce ambiguity and improve reviewer matching. Our dataset, methods, and findings offer a reproducible baseline for area selection support in ACL workflows.
%R 10.63317/4jbm8o3us4c8
%U https://aclanthology.org/2026.nslp-1.2/
%U https://doi.org/10.63317/4jbm8o3us4c8
%P 13-24
Markdown (Informal)
[Benchmarking LLMs for ARR Area Assignment: Evidence and Implications for Assignment Strategies](https://aclanthology.org/2026.nslp-1.2/) (Bingert et al., NSLP 2026)
ACL