@inproceedings{hasanov-ezzini-2026-judging,
title = "Judging the {LLM} Judges: A Human-Centric Validation of {LLM}-Generated Training Data for Software Retrieval",
author = "Hasanov, Ogtay and
Ezzini, Saad",
editor = "Mitkov, Ruslan and
Mu{\~n}oz, Rafael and
Lloret, Elena and
Ranasinghe, Tharindu and
Estevanell-Valladares, Ernesto L. and
Lamsiyah, Salima and
Montoyo, Andr{\'e}s and
Ezzini, Saad",
booktitle = "Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security",
month = jun,
year = "2026",
address = "Alicante, Spain",
publisher = "Department of Languages and Information Systems, University of Alicante",
url = "https://aclanthology.org/2026.nlpaics-1.20/",
pages = "184--188",
abstract = "As Large Language Models (LLMs) increasingly generate training data for downstream machine learning systems, the quality of this synthetic data becomes a critical security concern. Low-quality synthetic training data can silently poison retrieval systems deployed in security-sensitive contexts such as software issue triage, user support, and threat intelligence matching. We present a multi-dimensional quality assessment protocol for LLM-generated synthetic training data and apply it to a case study involving 13{,}579 synthetic user reviews generated from GitHub issues across four open-source Android applications. We evaluate 400 stratified samples using an LLM judge (GPT-4o-mini) along a five-point rubric, find that 10.5{\%} of generated reviews fail to meaningfully capture their source issues, and identify systematic failure patterns concentrated in developer-internal issues (continuous integration, refactoring) and sarcastic persona framings. To validate the LLM judge against human annotation, we compute Cohen{'}s Kappa between one human rater, a second independent human rater, and the LLM judge on 20 stratified reviews. Our results highlight the need for hybrid human-AI protocols when assessing synthetic data quality for security-critical applications."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hasanov-ezzini-2026-judging">
<titleInfo>
<title>Judging the LLM Judges: A Human-Centric Validation of LLM-Generated Training Data for Software Retrieval</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ogtay</namePart>
<namePart type="family">Hasanov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ruslan</namePart>
<namePart type="family">Mitkov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rafael</namePart>
<namePart type="family">Muñoz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Lloret</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tharindu</namePart>
<namePart type="family">Ranasinghe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ernesto</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Estevanell-Valladares</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Salima</namePart>
<namePart type="family">Lamsiyah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andrés</namePart>
<namePart type="family">Montoyo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Department of Languages and Information Systems, University of Alicante</publisher>
<place>
<placeTerm type="text">Alicante, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>As Large Language Models (LLMs) increasingly generate training data for downstream machine learning systems, the quality of this synthetic data becomes a critical security concern. Low-quality synthetic training data can silently poison retrieval systems deployed in security-sensitive contexts such as software issue triage, user support, and threat intelligence matching. We present a multi-dimensional quality assessment protocol for LLM-generated synthetic training data and apply it to a case study involving 13,579 synthetic user reviews generated from GitHub issues across four open-source Android applications. We evaluate 400 stratified samples using an LLM judge (GPT-4o-mini) along a five-point rubric, find that 10.5% of generated reviews fail to meaningfully capture their source issues, and identify systematic failure patterns concentrated in developer-internal issues (continuous integration, refactoring) and sarcastic persona framings. To validate the LLM judge against human annotation, we compute Cohen’s Kappa between one human rater, a second independent human rater, and the LLM judge on 20 stratified reviews. Our results highlight the need for hybrid human-AI protocols when assessing synthetic data quality for security-critical applications.</abstract>
<identifier type="citekey">hasanov-ezzini-2026-judging</identifier>
<location>
<url>https://aclanthology.org/2026.nlpaics-1.20/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>184</start>
<end>188</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Judging the LLM Judges: A Human-Centric Validation of LLM-Generated Training Data for Software Retrieval
%A Hasanov, Ogtay
%A Ezzini, Saad
%Y Mitkov, Ruslan
%Y Muñoz, Rafael
%Y Lloret, Elena
%Y Ranasinghe, Tharindu
%Y Estevanell-Valladares, Ernesto L.
%Y Lamsiyah, Salima
%Y Montoyo, Andrés
%Y Ezzini, Saad
%S Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security
%D 2026
%8 June
%I Department of Languages and Information Systems, University of Alicante
%C Alicante, Spain
%F hasanov-ezzini-2026-judging
%X As Large Language Models (LLMs) increasingly generate training data for downstream machine learning systems, the quality of this synthetic data becomes a critical security concern. Low-quality synthetic training data can silently poison retrieval systems deployed in security-sensitive contexts such as software issue triage, user support, and threat intelligence matching. We present a multi-dimensional quality assessment protocol for LLM-generated synthetic training data and apply it to a case study involving 13,579 synthetic user reviews generated from GitHub issues across four open-source Android applications. We evaluate 400 stratified samples using an LLM judge (GPT-4o-mini) along a five-point rubric, find that 10.5% of generated reviews fail to meaningfully capture their source issues, and identify systematic failure patterns concentrated in developer-internal issues (continuous integration, refactoring) and sarcastic persona framings. To validate the LLM judge against human annotation, we compute Cohen’s Kappa between one human rater, a second independent human rater, and the LLM judge on 20 stratified reviews. Our results highlight the need for hybrid human-AI protocols when assessing synthetic data quality for security-critical applications.
%U https://aclanthology.org/2026.nlpaics-1.20/
%P 184-188
Markdown (Informal)
[Judging the LLM Judges: A Human-Centric Validation of LLM-Generated Training Data for Software Retrieval](https://aclanthology.org/2026.nlpaics-1.20/) (Hasanov & Ezzini, NLPAICS 2026)
ACL