@inproceedings{wang-etal-2026-responsible,
title = "Responsible use of generative {AI} when creating reading comprehension questions: Inference matters",
author = "Wang, Zuowei and
Flor, Michael and
Zu, Jiyun and
O{'}Reilly, Tenaha and
Ma, Wanjing Anya",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.27/",
pages = "207--210",
ISBN = "979-8-9983004-1-7",
abstract = "AI-generated and expert-created reading comprehension questions can show similar item statistics yet differ in the types of inferences required. This difference stemmed from AI{'}s failure to follow prompts during an intermediate item generation step. Evaluations of AI-generated items should document prompts and generation steps to identify and mitigate construct-relevant differences."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="wang-etal-2026-responsible">
<titleInfo>
<title>Responsible use of generative AI when creating reading comprehension questions: Inference matters</title>
</titleInfo>
<name type="personal">
<namePart type="given">Zuowei</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Michael</namePart>
<namePart type="family">Flor</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jiyun</namePart>
<namePart type="family">Zu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tenaha</namePart>
<namePart type="family">O’Reilly</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Wanjing</namePart>
<namePart type="given">Anya</namePart>
<namePart type="family">Ma</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>AI-generated and expert-created reading comprehension questions can show similar item statistics yet differ in the types of inferences required. This difference stemmed from AI’s failure to follow prompts during an intermediate item generation step. Evaluations of AI-generated items should document prompts and generation steps to identify and mitigate construct-relevant differences.</abstract>
<identifier type="citekey">wang-etal-2026-responsible</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.27/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>207</start>
<end>210</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Responsible use of generative AI when creating reading comprehension questions: Inference matters
%A Wang, Zuowei
%A Flor, Michael
%A Zu, Jiyun
%A O’Reilly, Tenaha
%A Ma, Wanjing Anya
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F wang-etal-2026-responsible
%X AI-generated and expert-created reading comprehension questions can show similar item statistics yet differ in the types of inferences required. This difference stemmed from AI’s failure to follow prompts during an intermediate item generation step. Evaluations of AI-generated items should document prompts and generation steps to identify and mitigate construct-relevant differences.
%U https://aclanthology.org/2026.aimecon-wip.27/
%P 207-210
Markdown (Informal)
[Responsible use of generative AI when creating reading comprehension questions: Inference matters](https://aclanthology.org/2026.aimecon-wip.27/) (Wang et al., AIME-Con 2026)
ACL