@inproceedings{yao-etal-2026-evaluation,
title = "An Evaluation of Agreement and Uncertainty in {LLM}-Based Automated Essay Scoring",
author = "Yao, Yiting and
Yang, Yanyun and
Kuang, Huan (Hailey)",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.31/",
pages = "280--287",
ISBN = "979-8-9983004-0-0",
abstract = "This study evaluates score agreement and uncertainty estimation in large language model (LLM)-based automated essay scoring. Results showed that LLMs tended to be overconfident and highly consistent in their assigned scores, and majority voting did not yield stronger agreement with human scores than deterministic scoring."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="yao-etal-2026-evaluation">
<titleInfo>
<title>An Evaluation of Agreement and Uncertainty in LLM-Based Automated Essay Scoring</title>
</titleInfo>
<name type="personal">
<namePart type="given">Yiting</namePart>
<namePart type="family">Yao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yanyun</namePart>
<namePart type="family">Yang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Huan</namePart>
<namePart type="given">(Hailey)</namePart>
<namePart type="family">Kuang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>This study evaluates score agreement and uncertainty estimation in large language model (LLM)-based automated essay scoring. Results showed that LLMs tended to be overconfident and highly consistent in their assigned scores, and majority voting did not yield stronger agreement with human scores than deterministic scoring.</abstract>
<identifier type="citekey">yao-etal-2026-evaluation</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.31/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>280</start>
<end>287</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T An Evaluation of Agreement and Uncertainty in LLM-Based Automated Essay Scoring
%A Yao, Yiting
%A Yang, Yanyun
%A Kuang, Huan (Hailey)
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F yao-etal-2026-evaluation
%X This study evaluates score agreement and uncertainty estimation in large language model (LLM)-based automated essay scoring. Results showed that LLMs tended to be overconfident and highly consistent in their assigned scores, and majority voting did not yield stronger agreement with human scores than deterministic scoring.
%U https://aclanthology.org/2026.aimecon-main.31/
%P 280-287
Markdown (Informal)
[An Evaluation of Agreement and Uncertainty in LLM-Based Automated Essay Scoring](https://aclanthology.org/2026.aimecon-main.31/) (Yao et al., AIME-Con 2026)
ACL