@inproceedings{chen-etal-2026-using,
title = "Using Generative {AI} to Simulate Item Responses by Skills Insight Score Band",
author = "Chen, Jianshen and
Lee, Chansoon and
Gorney, Kylie and
Antal, Judit",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.18/",
pages = "140--144",
ISBN = "979-8-9983004-1-7",
abstract = "This study evaluates GPT-5.4 Thinking simulations of reading-item responses using Skills Insight score-band descriptions and item content. Simulated and empirical item statistics and IRT parameters were compared. Difficulty recovery was strongest, with moderately strong b-parameter correlations, while a and c recovery was weaker, supporting preliminary difficulty evaluation before field testing."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="chen-etal-2026-using">
<titleInfo>
<title>Using Generative AI to Simulate Item Responses by Skills Insight Score Band</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jianshen</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chansoon</namePart>
<namePart type="family">Lee</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kylie</namePart>
<namePart type="family">Gorney</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Judit</namePart>
<namePart type="family">Antal</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>This study evaluates GPT-5.4 Thinking simulations of reading-item responses using Skills Insight score-band descriptions and item content. Simulated and empirical item statistics and IRT parameters were compared. Difficulty recovery was strongest, with moderately strong b-parameter correlations, while a and c recovery was weaker, supporting preliminary difficulty evaluation before field testing.</abstract>
<identifier type="citekey">chen-etal-2026-using</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.18/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>140</start>
<end>144</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Using Generative AI to Simulate Item Responses by Skills Insight Score Band
%A Chen, Jianshen
%A Lee, Chansoon
%A Gorney, Kylie
%A Antal, Judit
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F chen-etal-2026-using
%X This study evaluates GPT-5.4 Thinking simulations of reading-item responses using Skills Insight score-band descriptions and item content. Simulated and empirical item statistics and IRT parameters were compared. Difficulty recovery was strongest, with moderately strong b-parameter correlations, while a and c recovery was weaker, supporting preliminary difficulty evaluation before field testing.
%U https://aclanthology.org/2026.aimecon-wip.18/
%P 140-144
Markdown (Informal)
[Using Generative AI to Simulate Item Responses by Skills Insight Score Band](https://aclanthology.org/2026.aimecon-wip.18/) (Chen et al., AIME-Con 2026)
ACL
- Jianshen Chen, Chansoon Lee, Kylie Gorney, and Judit Antal. 2026. Using Generative AI to Simulate Item Responses by Skills Insight Score Band. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress, pages 140–144, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).