@inproceedings{ma-2026-validation,
title = "From Validation to Resilience: Sustaining Measurement Quality in {AI}-Assisted Enemy Item Identification",
author = "Ma, Ye",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Coordinated Session Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-sessions.9/",
pages = "76--81",
ISBN = "979-8-9983004-2-4",
abstract = "Enemy items are item pairs that must not appear on the same test form. An automatic enemy-identification method using large language models (LLMs) has been deployed in operation. This study asks a question: how is measurement quality sustained as conditions shift after deployment? Using operational data from certification exams, two analyses examine the factors that affect the method{'}s resilience. Analysis 1 isolates model-version and prompt updates: changing the model with the prompt held constant reduced recall from 0.75 to 0.54, while prompt refinement with the model held constant recovered it to 0.82. It also shows that most model{--}reviewer disagreements are edge cases and that the human standard is itself variable, with four experts spanning 0.70 to 0.91 in recall. Analysis 2 reports a disruption case: the established method works effectively on a professional-level exam but not on a foundational-level exam. A complementary content-tag based method was added to the existing method in response, with human review surfacing the disruption and validating the fix. Responsible LLM deployment in assessment requires monitoring with labeled data, testing before deployment, and evaluation systems tied to each use case, so that validity, reliability, and fairness are sustained rather than certified once."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="ma-2026-validation">
<titleInfo>
<title>From Validation to Resilience: Sustaining Measurement Quality in AI-Assisted Enemy Item Identification</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ye</namePart>
<namePart type="family">Ma</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-2-4</identifier>
</relatedItem>
<abstract>Enemy items are item pairs that must not appear on the same test form. An automatic enemy-identification method using large language models (LLMs) has been deployed in operation. This study asks a question: how is measurement quality sustained as conditions shift after deployment? Using operational data from certification exams, two analyses examine the factors that affect the method’s resilience. Analysis 1 isolates model-version and prompt updates: changing the model with the prompt held constant reduced recall from 0.75 to 0.54, while prompt refinement with the model held constant recovered it to 0.82. It also shows that most model–reviewer disagreements are edge cases and that the human standard is itself variable, with four experts spanning 0.70 to 0.91 in recall. Analysis 2 reports a disruption case: the established method works effectively on a professional-level exam but not on a foundational-level exam. A complementary content-tag based method was added to the existing method in response, with human review surfacing the disruption and validating the fix. Responsible LLM deployment in assessment requires monitoring with labeled data, testing before deployment, and evaluation systems tied to each use case, so that validity, reliability, and fairness are sustained rather than certified once.</abstract>
<identifier type="citekey">ma-2026-validation</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-sessions.9/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>76</start>
<end>81</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T From Validation to Resilience: Sustaining Measurement Quality in AI-Assisted Enemy Item Identification
%A Ma, Ye
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-2-4
%F ma-2026-validation
%X Enemy items are item pairs that must not appear on the same test form. An automatic enemy-identification method using large language models (LLMs) has been deployed in operation. This study asks a question: how is measurement quality sustained as conditions shift after deployment? Using operational data from certification exams, two analyses examine the factors that affect the method’s resilience. Analysis 1 isolates model-version and prompt updates: changing the model with the prompt held constant reduced recall from 0.75 to 0.54, while prompt refinement with the model held constant recovered it to 0.82. It also shows that most model–reviewer disagreements are edge cases and that the human standard is itself variable, with four experts spanning 0.70 to 0.91 in recall. Analysis 2 reports a disruption case: the established method works effectively on a professional-level exam but not on a foundational-level exam. A complementary content-tag based method was added to the existing method in response, with human review surfacing the disruption and validating the fix. Responsible LLM deployment in assessment requires monitoring with labeled data, testing before deployment, and evaluation systems tied to each use case, so that validity, reliability, and fairness are sustained rather than certified once.
%U https://aclanthology.org/2026.aimecon-sessions.9/
%P 76-81
Markdown (Informal)
[From Validation to Resilience: Sustaining Measurement Quality in AI-Assisted Enemy Item Identification](https://aclanthology.org/2026.aimecon-sessions.9/) (Ma, AIME-Con 2026)
ACL