@inproceedings{safar-etal-2026-evaluating,
title = "Evaluating an {AI}-Item Generation Tool for {NAEP} Science",
author = "Safar, Carolina and
Price, Mitchell and
Ackley, Jeffery",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Coordinated Session Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-sessions.21/",
pages = "203--209",
ISBN = "979-8-9983004-2-4",
abstract = "AI-enabled systems could transform item authoring for standardized assessments. We compare framework alignment and accuracy of NAEP Science items developed by two AI tools and human authors. Items from all sources had similar acceptance rates and most accepted items would require significant revision, underscoring the importance of evaluating AI-generated content."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="safar-etal-2026-evaluating">
<titleInfo>
<title>Evaluating an AI-Item Generation Tool for NAEP Science</title>
</titleInfo>
<name type="personal">
<namePart type="given">Carolina</namePart>
<namePart type="family">Safar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mitchell</namePart>
<namePart type="family">Price</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jeffery</namePart>
<namePart type="family">Ackley</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-2-4</identifier>
</relatedItem>
<abstract>AI-enabled systems could transform item authoring for standardized assessments. We compare framework alignment and accuracy of NAEP Science items developed by two AI tools and human authors. Items from all sources had similar acceptance rates and most accepted items would require significant revision, underscoring the importance of evaluating AI-generated content.</abstract>
<identifier type="citekey">safar-etal-2026-evaluating</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-sessions.21/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>203</start>
<end>209</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Evaluating an AI-Item Generation Tool for NAEP Science
%A Safar, Carolina
%A Price, Mitchell
%A Ackley, Jeffery
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-2-4
%F safar-etal-2026-evaluating
%X AI-enabled systems could transform item authoring for standardized assessments. We compare framework alignment and accuracy of NAEP Science items developed by two AI tools and human authors. Items from all sources had similar acceptance rates and most accepted items would require significant revision, underscoring the importance of evaluating AI-generated content.
%U https://aclanthology.org/2026.aimecon-sessions.21/
%P 203-209
Markdown (Informal)
[Evaluating an AI-Item Generation Tool for NAEP Science](https://aclanthology.org/2026.aimecon-sessions.21/) (Safar et al., AIME-Con 2026)
ACL
- Carolina Safar, Mitchell Price, and Jeffery Ackley. 2026. Evaluating an AI-Item Generation Tool for NAEP Science. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers, pages 203–209, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).