@inproceedings{moses-etal-2026-score,
title = "Score Comparability Limits with Common {AES} Validation Measures",
author = "Moses, Tim and
Kim, Seongeun and
Trout, Nick",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.13/",
pages = "93--99",
ISBN = "979-8-9983004-1-7",
abstract = "This study considers measures commonly used to validate AES scoring and their limitations for indicating comparability with human rater scoring. In data simulated to reflect validation results achieved by AES national competition winners, scoring standard differences can occur across AES and human rater scoring (especially for 4-point scales vs. 2- and 3-point scales). Equipercentile methods are described and recommended for resolving the scoring differences."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="moses-etal-2026-score">
<titleInfo>
<title>Score Comparability Limits with Common AES Validation Measures</title>
</titleInfo>
<name type="personal">
<namePart type="given">Tim</namePart>
<namePart type="family">Moses</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Seongeun</namePart>
<namePart type="family">Kim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nick</namePart>
<namePart type="family">Trout</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>This study considers measures commonly used to validate AES scoring and their limitations for indicating comparability with human rater scoring. In data simulated to reflect validation results achieved by AES national competition winners, scoring standard differences can occur across AES and human rater scoring (especially for 4-point scales vs. 2- and 3-point scales). Equipercentile methods are described and recommended for resolving the scoring differences.</abstract>
<identifier type="citekey">moses-etal-2026-score</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.13/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>93</start>
<end>99</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Score Comparability Limits with Common AES Validation Measures
%A Moses, Tim
%A Kim, Seongeun
%A Trout, Nick
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F moses-etal-2026-score
%X This study considers measures commonly used to validate AES scoring and their limitations for indicating comparability with human rater scoring. In data simulated to reflect validation results achieved by AES national competition winners, scoring standard differences can occur across AES and human rater scoring (especially for 4-point scales vs. 2- and 3-point scales). Equipercentile methods are described and recommended for resolving the scoring differences.
%U https://aclanthology.org/2026.aimecon-wip.13/
%P 93-99
Markdown (Informal)
[Score Comparability Limits with Common AES Validation Measures](https://aclanthology.org/2026.aimecon-wip.13/) (Moses et al., AIME-Con 2026)
ACL
- Tim Moses, Seongeun Kim, and Nick Trout. 2026. Score Comparability Limits with Common AES Validation Measures. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress, pages 93–99, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).