@inproceedings{perez-etal-2026-evaluating,
title = "Evaluating Multiple Models for Predicting Item Difficulty in a Principled Assessment Design",
author = "Perez, Alexandra Lane and
Schneider, Christina and
Lim, Sangdon and
Gianopulos, Garron and
Xue, Kang",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.20/",
pages = "152--158",
ISBN = "979-8-9983004-1-7",
abstract = "Item difficulty should, in theory, be predictable from features derived from RPLDs, task characteristics, and linguistic complexity. This study evaluates the performance of multiple statistical and machine learning models in estimating item difficulty. We found similar results across all four models, the item features selected explain 53{\%}{--}56{\%} of the variance across all grades."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="perez-etal-2026-evaluating">
<titleInfo>
<title>Evaluating Multiple Models for Predicting Item Difficulty in a Principled Assessment Design</title>
</titleInfo>
<name type="personal">
<namePart type="given">Alexandra</namePart>
<namePart type="given">Lane</namePart>
<namePart type="family">Perez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christina</namePart>
<namePart type="family">Schneider</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sangdon</namePart>
<namePart type="family">Lim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Garron</namePart>
<namePart type="family">Gianopulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kang</namePart>
<namePart type="family">Xue</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>Item difficulty should, in theory, be predictable from features derived from RPLDs, task characteristics, and linguistic complexity. This study evaluates the performance of multiple statistical and machine learning models in estimating item difficulty. We found similar results across all four models, the item features selected explain 53%–56% of the variance across all grades.</abstract>
<identifier type="citekey">perez-etal-2026-evaluating</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.20/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>152</start>
<end>158</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Evaluating Multiple Models for Predicting Item Difficulty in a Principled Assessment Design
%A Perez, Alexandra Lane
%A Schneider, Christina
%A Lim, Sangdon
%A Gianopulos, Garron
%A Xue, Kang
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F perez-etal-2026-evaluating
%X Item difficulty should, in theory, be predictable from features derived from RPLDs, task characteristics, and linguistic complexity. This study evaluates the performance of multiple statistical and machine learning models in estimating item difficulty. We found similar results across all four models, the item features selected explain 53%–56% of the variance across all grades.
%U https://aclanthology.org/2026.aimecon-wip.20/
%P 152-158
Markdown (Informal)
[Evaluating Multiple Models for Predicting Item Difficulty in a Principled Assessment Design](https://aclanthology.org/2026.aimecon-wip.20/) (Perez et al., AIME-Con 2026)
ACL