@inproceedings{dicerbo-etal-2026-applying,
title = "Applying Evidence-Centered Design to Automated Evals of {AI}-Powered Assessment Systems",
author = "DiCerbo, Kristen and
Cheng, Britte Haugan and
Whitmer, John",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.42/",
pages = "329--334",
ISBN = "979-8-9983004-1-7",
abstract = "This paper demonstrates an application of Evidence-Centered Design (ECD) as a principled approach to design automated evaluations of AI-powered assessment outputs. We demonstrate this application through Khan Academy{'}s ``Explain Your Thinking'' conversational agent for mathematics items, showing how ECD{'}s layered models can be translated into rigorous and interpretable systems to demonstrate validity, reliability and fairness in AI systems to diverse audiences."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="dicerbo-etal-2026-applying">
<titleInfo>
<title>Applying Evidence-Centered Design to Automated Evals of AI-Powered Assessment Systems</title>
</titleInfo>
<name type="personal">
<namePart type="given">Kristen</namePart>
<namePart type="family">DiCerbo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Britte</namePart>
<namePart type="given">Haugan</namePart>
<namePart type="family">Cheng</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">John</namePart>
<namePart type="family">Whitmer</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>This paper demonstrates an application of Evidence-Centered Design (ECD) as a principled approach to design automated evaluations of AI-powered assessment outputs. We demonstrate this application through Khan Academy’s “Explain Your Thinking” conversational agent for mathematics items, showing how ECD’s layered models can be translated into rigorous and interpretable systems to demonstrate validity, reliability and fairness in AI systems to diverse audiences.</abstract>
<identifier type="citekey">dicerbo-etal-2026-applying</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.42/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>329</start>
<end>334</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Applying Evidence-Centered Design to Automated Evals of AI-Powered Assessment Systems
%A DiCerbo, Kristen
%A Cheng, Britte Haugan
%A Whitmer, John
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F dicerbo-etal-2026-applying
%X This paper demonstrates an application of Evidence-Centered Design (ECD) as a principled approach to design automated evaluations of AI-powered assessment outputs. We demonstrate this application through Khan Academy’s “Explain Your Thinking” conversational agent for mathematics items, showing how ECD’s layered models can be translated into rigorous and interpretable systems to demonstrate validity, reliability and fairness in AI systems to diverse audiences.
%U https://aclanthology.org/2026.aimecon-wip.42/
%P 329-334
Markdown (Informal)
[Applying Evidence-Centered Design to Automated Evals of AI-Powered Assessment Systems](https://aclanthology.org/2026.aimecon-wip.42/) (DiCerbo et al., AIME-Con 2026)
ACL