@inproceedings{han-rijmen-2026-llm,
title = "{LLM}-Based Pairwise Judgment for Math Item Parameter Modeling Using Workflows and Agents",
author = "Han, Suhwa and
Rijmen, Frank",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.32/",
pages = "248--259",
ISBN = "979-8-9983004-1-7",
abstract = "This study examines the feasibility of using large language models (LLMs) as judges for pairwise comparisons of item difficulty and to the extent which the resulting comparison outcomes recover banked difficulty parameters. The study in particular fine-tunes an instruction-tuned LLM to evaluate the impact of task-specific fine-tuning on the parameter prediction accuracy. The study also demonstrates an agentic approach to hyperparameter tuning of LLM-training task."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="han-rijmen-2026-llm">
<titleInfo>
<title>LLM-Based Pairwise Judgment for Math Item Parameter Modeling Using Workflows and Agents</title>
</titleInfo>
<name type="personal">
<namePart type="given">Suhwa</namePart>
<namePart type="family">Han</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Frank</namePart>
<namePart type="family">Rijmen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>This study examines the feasibility of using large language models (LLMs) as judges for pairwise comparisons of item difficulty and to the extent which the resulting comparison outcomes recover banked difficulty parameters. The study in particular fine-tunes an instruction-tuned LLM to evaluate the impact of task-specific fine-tuning on the parameter prediction accuracy. The study also demonstrates an agentic approach to hyperparameter tuning of LLM-training task.</abstract>
<identifier type="citekey">han-rijmen-2026-llm</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.32/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>248</start>
<end>259</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T LLM-Based Pairwise Judgment for Math Item Parameter Modeling Using Workflows and Agents
%A Han, Suhwa
%A Rijmen, Frank
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F han-rijmen-2026-llm
%X This study examines the feasibility of using large language models (LLMs) as judges for pairwise comparisons of item difficulty and to the extent which the resulting comparison outcomes recover banked difficulty parameters. The study in particular fine-tunes an instruction-tuned LLM to evaluate the impact of task-specific fine-tuning on the parameter prediction accuracy. The study also demonstrates an agentic approach to hyperparameter tuning of LLM-training task.
%U https://aclanthology.org/2026.aimecon-wip.32/
%P 248-259
Markdown (Informal)
[LLM-Based Pairwise Judgment for Math Item Parameter Modeling Using Workflows and Agents](https://aclanthology.org/2026.aimecon-wip.32/) (Han & Rijmen, AIME-Con 2026)
ACL