@inproceedings{li-etal-2026-using,
title = "Using {LLM} Judges' Paired Comparisons to Estimate Mathematics Item Difficulty",
author = "Li, Lanrong and
Abulela, Mohammad A. A. and
Gushta, Matthew",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.24/",
pages = "184--192",
ISBN = "979-8-9983004-1-7",
abstract = "Estimating item difficulty without collecting field testing data has been a long sought-after goal in educational measurement. We prompted large language models (LLMs) to compare mathematics items in pairs to estimate their difficulty. Results showed strong correlations between difficulty based on one LLM judge{'}s paired comparisons and empirical item difficulty."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="li-etal-2026-using">
<titleInfo>
<title>Using LLM Judges’ Paired Comparisons to Estimate Mathematics Item Difficulty</title>
</titleInfo>
<name type="personal">
<namePart type="given">Lanrong</namePart>
<namePart type="family">Li</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mohammad</namePart>
<namePart type="given">A</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Abulela</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Matthew</namePart>
<namePart type="family">Gushta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>Estimating item difficulty without collecting field testing data has been a long sought-after goal in educational measurement. We prompted large language models (LLMs) to compare mathematics items in pairs to estimate their difficulty. Results showed strong correlations between difficulty based on one LLM judge’s paired comparisons and empirical item difficulty.</abstract>
<identifier type="citekey">li-etal-2026-using</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.24/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>184</start>
<end>192</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Using LLM Judges’ Paired Comparisons to Estimate Mathematics Item Difficulty
%A Li, Lanrong
%A Abulela, Mohammad A. A.
%A Gushta, Matthew
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F li-etal-2026-using
%X Estimating item difficulty without collecting field testing data has been a long sought-after goal in educational measurement. We prompted large language models (LLMs) to compare mathematics items in pairs to estimate their difficulty. Results showed strong correlations between difficulty based on one LLM judge’s paired comparisons and empirical item difficulty.
%U https://aclanthology.org/2026.aimecon-wip.24/
%P 184-192
Markdown (Informal)
[Using LLM Judges’ Paired Comparisons to Estimate Mathematics Item Difficulty](https://aclanthology.org/2026.aimecon-wip.24/) (Li et al., AIME-Con 2026)
ACL
- Lanrong Li, Mohammad A. A. Abulela, and Matthew Gushta. 2026. Using LLM Judges’ Paired Comparisons to Estimate Mathematics Item Difficulty. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress, pages 184–192, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).