@inproceedings{poudel-etal-2026-two,
title = "Two Uses of Human-scored Anchors: Few-shot Direct Scoring and Anchor-based Comparative Grading",
author = "Poudel, Prashreet and
Gu, Huayue and
Lynch, Collin and
Gao, Zhikai",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.51/",
pages = "458--465",
ISBN = "979-8-9983004-0-0",
abstract = "This study evaluates direct few-shot scoring and anchor based comparative judgment for LLM essay grading. Using human scored anchor essays, we compare their accuracy and stability. Although both approaches produce competitive scores, comparative judgments demonstrate better stability and consistency, suggesting it as a better alternative for grading writing assessment."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="poudel-etal-2026-two">
<titleInfo>
<title>Two Uses of Human-scored Anchors: Few-shot Direct Scoring and Anchor-based Comparative Grading</title>
</titleInfo>
<name type="personal">
<namePart type="given">Prashreet</namePart>
<namePart type="family">Poudel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Huayue</namePart>
<namePart type="family">Gu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Collin</namePart>
<namePart type="family">Lynch</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Zhikai</namePart>
<namePart type="family">Gao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>This study evaluates direct few-shot scoring and anchor based comparative judgment for LLM essay grading. Using human scored anchor essays, we compare their accuracy and stability. Although both approaches produce competitive scores, comparative judgments demonstrate better stability and consistency, suggesting it as a better alternative for grading writing assessment.</abstract>
<identifier type="citekey">poudel-etal-2026-two</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.51/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>458</start>
<end>465</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Two Uses of Human-scored Anchors: Few-shot Direct Scoring and Anchor-based Comparative Grading
%A Poudel, Prashreet
%A Gu, Huayue
%A Lynch, Collin
%A Gao, Zhikai
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F poudel-etal-2026-two
%X This study evaluates direct few-shot scoring and anchor based comparative judgment for LLM essay grading. Using human scored anchor essays, we compare their accuracy and stability. Although both approaches produce competitive scores, comparative judgments demonstrate better stability and consistency, suggesting it as a better alternative for grading writing assessment.
%U https://aclanthology.org/2026.aimecon-main.51/
%P 458-465
Markdown (Informal)
[Two Uses of Human-scored Anchors: Few-shot Direct Scoring and Anchor-based Comparative Grading](https://aclanthology.org/2026.aimecon-main.51/) (Poudel et al., AIME-Con 2026)
ACL