@inproceedings{jewsbury-etal-2026-bayesian,
title = "{B}ayesian Consensus Calibration of Continuously Evolving {IRT} Item Banks",
author = "Jewsbury, Paul A and
Nydick, Steven W and
Liao, Manqian and
Chen, Siyuan (Marco)",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.35/",
pages = "314--322",
ISBN = "979-8-9983004-0-0",
abstract = "AI-based item generation and NLP-based prediction of item parameters are producing item banks that are substantially larger, sparser, and more frequently updated than conventional banks. Hierarchical Bayesian item response theory (IRT) is a natural calibration framework for such banks, but the common practice of refitting the entire accumulated response history at each update is costly and can exceed available memory. We describe consensus calibration, a divide-and-conquer procedure that calibrates each time period independently and reconstructs the pooled posterior in two layers. First, the posterior draws of each period are mapped to a common metric by a robust characteristic-curve linking (Haebara) that is solved separately for each draw, which propagates the uncertainty of the linking transformation into the linked posteriors. Second, the linked item posteriors are combined as a product of Gaussian densities from which the population prior contributed by each period is removed and a single prior{---}obtained by consensus across the per-period population posteriors{---}is reinstated. The correction targets the posterior dispersion, not only its location. As evidence for consensus calibration, we compare it to a pooled single-run analysis on a large operational assessment in terms of item-parameter recovery, an uncertainty-by-exposure diagnostic, and the ability distributions."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="jewsbury-etal-2026-bayesian">
<titleInfo>
<title>Bayesian Consensus Calibration of Continuously Evolving IRT Item Banks</title>
</titleInfo>
<name type="personal">
<namePart type="given">Paul</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Jewsbury</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Steven</namePart>
<namePart type="given">W</namePart>
<namePart type="family">Nydick</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Manqian</namePart>
<namePart type="family">Liao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Siyuan</namePart>
<namePart type="given">(Marco)</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>AI-based item generation and NLP-based prediction of item parameters are producing item banks that are substantially larger, sparser, and more frequently updated than conventional banks. Hierarchical Bayesian item response theory (IRT) is a natural calibration framework for such banks, but the common practice of refitting the entire accumulated response history at each update is costly and can exceed available memory. We describe consensus calibration, a divide-and-conquer procedure that calibrates each time period independently and reconstructs the pooled posterior in two layers. First, the posterior draws of each period are mapped to a common metric by a robust characteristic-curve linking (Haebara) that is solved separately for each draw, which propagates the uncertainty of the linking transformation into the linked posteriors. Second, the linked item posteriors are combined as a product of Gaussian densities from which the population prior contributed by each period is removed and a single prior—obtained by consensus across the per-period population posteriors—is reinstated. The correction targets the posterior dispersion, not only its location. As evidence for consensus calibration, we compare it to a pooled single-run analysis on a large operational assessment in terms of item-parameter recovery, an uncertainty-by-exposure diagnostic, and the ability distributions.</abstract>
<identifier type="citekey">jewsbury-etal-2026-bayesian</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.35/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>314</start>
<end>322</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Bayesian Consensus Calibration of Continuously Evolving IRT Item Banks
%A Jewsbury, Paul A.
%A Nydick, Steven W.
%A Liao, Manqian
%A Chen, Siyuan (Marco)
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F jewsbury-etal-2026-bayesian
%X AI-based item generation and NLP-based prediction of item parameters are producing item banks that are substantially larger, sparser, and more frequently updated than conventional banks. Hierarchical Bayesian item response theory (IRT) is a natural calibration framework for such banks, but the common practice of refitting the entire accumulated response history at each update is costly and can exceed available memory. We describe consensus calibration, a divide-and-conquer procedure that calibrates each time period independently and reconstructs the pooled posterior in two layers. First, the posterior draws of each period are mapped to a common metric by a robust characteristic-curve linking (Haebara) that is solved separately for each draw, which propagates the uncertainty of the linking transformation into the linked posteriors. Second, the linked item posteriors are combined as a product of Gaussian densities from which the population prior contributed by each period is removed and a single prior—obtained by consensus across the per-period population posteriors—is reinstated. The correction targets the posterior dispersion, not only its location. As evidence for consensus calibration, we compare it to a pooled single-run analysis on a large operational assessment in terms of item-parameter recovery, an uncertainty-by-exposure diagnostic, and the ability distributions.
%U https://aclanthology.org/2026.aimecon-main.35/
%P 314-322
Markdown (Informal)
[Bayesian Consensus Calibration of Continuously Evolving IRT Item Banks](https://aclanthology.org/2026.aimecon-main.35/) (Jewsbury et al., AIME-Con 2026)
ACL
- Paul A Jewsbury, Steven W Nydick, Manqian Liao, and Siyuan (Marco) Chen. 2026. Bayesian Consensus Calibration of Continuously Evolving IRT Item Banks. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers, pages 314–322, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).