@inproceedings{hardy-2026-autoscoring,
title = "Autoscoring Anticlimax: A Meta-analytic Understanding of {AI}{'}s Short-answer Shortcomings and Wording Weaknesses",
author = "Hardy, Michael",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.13/",
pages = "117--138",
ISBN = "979-8-9983004-0-0",
abstract = "We meta-analyze 890 culminating results across a systematic review of LLM short-answer scoring studies. We estimate the factors that contribute to LLM performance, quantifying the disconnect between human and LLM difficulties in SAS and provide recommendations for developers working on SAS for schoolchildren."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hardy-2026-autoscoring">
<titleInfo>
<title>Autoscoring Anticlimax: A Meta-analytic Understanding of AI’s Short-answer Shortcomings and Wording Weaknesses</title>
</titleInfo>
<name type="personal">
<namePart type="given">Michael</namePart>
<namePart type="family">Hardy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>We meta-analyze 890 culminating results across a systematic review of LLM short-answer scoring studies. We estimate the factors that contribute to LLM performance, quantifying the disconnect between human and LLM difficulties in SAS and provide recommendations for developers working on SAS for schoolchildren.</abstract>
<identifier type="citekey">hardy-2026-autoscoring</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.13/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>117</start>
<end>138</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Autoscoring Anticlimax: A Meta-analytic Understanding of AI’s Short-answer Shortcomings and Wording Weaknesses
%A Hardy, Michael
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F hardy-2026-autoscoring
%X We meta-analyze 890 culminating results across a systematic review of LLM short-answer scoring studies. We estimate the factors that contribute to LLM performance, quantifying the disconnect between human and LLM difficulties in SAS and provide recommendations for developers working on SAS for schoolchildren.
%U https://aclanthology.org/2026.aimecon-main.13/
%P 117-138
Markdown (Informal)
[Autoscoring Anticlimax: A Meta-analytic Understanding of AI’s Short-answer Shortcomings and Wording Weaknesses](https://aclanthology.org/2026.aimecon-main.13/) (Hardy, AIME-Con 2026)
ACL