@inproceedings{kolle-etal-2026-composite,
title = "Composite Scores vs. Preference Rankings: Measuring Architecture Effects in {LLM} Feedback",
author = "Kolle, Harvey Ngoe and
Demmans Epp, Carrie and
Liaqat, Amna and
Cutumisu, Maria",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.30/",
pages = "231--234",
ISBN = "979-8-9983004-1-7",
abstract = "AI-generated feedback is often judged by quality ratings or preference rankings but rarely both. Comparing a multi-agent system, a single-agent system, and human feedback on student writing, we show that the two evaluation approaches can support different reported conclusions even when their underlying effects are nearly identical. These differences have consequences for how AI-generated feedback should be evaluated."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kolle-etal-2026-composite">
<titleInfo>
<title>Composite Scores vs. Preference Rankings: Measuring Architecture Effects in LLM Feedback</title>
</titleInfo>
<name type="personal">
<namePart type="given">Harvey</namePart>
<namePart type="given">Ngoe</namePart>
<namePart type="family">Kolle</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Carrie</namePart>
<namePart type="family">Demmans Epp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Amna</namePart>
<namePart type="family">Liaqat</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="family">Cutumisu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>AI-generated feedback is often judged by quality ratings or preference rankings but rarely both. Comparing a multi-agent system, a single-agent system, and human feedback on student writing, we show that the two evaluation approaches can support different reported conclusions even when their underlying effects are nearly identical. These differences have consequences for how AI-generated feedback should be evaluated.</abstract>
<identifier type="citekey">kolle-etal-2026-composite</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.30/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>231</start>
<end>234</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Composite Scores vs. Preference Rankings: Measuring Architecture Effects in LLM Feedback
%A Kolle, Harvey Ngoe
%A Demmans Epp, Carrie
%A Liaqat, Amna
%A Cutumisu, Maria
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F kolle-etal-2026-composite
%X AI-generated feedback is often judged by quality ratings or preference rankings but rarely both. Comparing a multi-agent system, a single-agent system, and human feedback on student writing, we show that the two evaluation approaches can support different reported conclusions even when their underlying effects are nearly identical. These differences have consequences for how AI-generated feedback should be evaluated.
%U https://aclanthology.org/2026.aimecon-wip.30/
%P 231-234
Markdown (Informal)
[Composite Scores vs. Preference Rankings: Measuring Architecture Effects in LLM Feedback](https://aclanthology.org/2026.aimecon-wip.30/) (Kolle et al., AIME-Con 2026)
ACL