@inproceedings{jung-etal-2026-semantic,
title = "Semantic Similarity is Not Enough for Comparing {LLM} and Human Rater Rationales",
author = {Jung, Julie Jongeun and
Lu, Max and
Darici, Dogus and
Br{\"u}egge, Emilia},
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.52/",
pages = "466--471",
ISBN = "979-8-9983004-0-0",
abstract = "Comparing LLM-human rater rationales using semantic similarity risks conflating textual proximity with evaluative agreement. We test whether embedding-based similarity reflects qualitative coding distinctions across rationale pairs in medical education. Similarity declined as rationale length differences grew and was less effective at distinguishing whether LLMs preserved the human{'}s central claim."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="jung-etal-2026-semantic">
<titleInfo>
<title>Semantic Similarity is Not Enough for Comparing LLM and Human Rater Rationales</title>
</titleInfo>
<name type="personal">
<namePart type="given">Julie</namePart>
<namePart type="given">Jongeun</namePart>
<namePart type="family">Jung</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Max</namePart>
<namePart type="family">Lu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dogus</namePart>
<namePart type="family">Darici</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Emilia</namePart>
<namePart type="family">Brüegge</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>Comparing LLM-human rater rationales using semantic similarity risks conflating textual proximity with evaluative agreement. We test whether embedding-based similarity reflects qualitative coding distinctions across rationale pairs in medical education. Similarity declined as rationale length differences grew and was less effective at distinguishing whether LLMs preserved the human’s central claim.</abstract>
<identifier type="citekey">jung-etal-2026-semantic</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.52/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>466</start>
<end>471</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Semantic Similarity is Not Enough for Comparing LLM and Human Rater Rationales
%A Jung, Julie Jongeun
%A Lu, Max
%A Darici, Dogus
%A Brüegge, Emilia
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F jung-etal-2026-semantic
%X Comparing LLM-human rater rationales using semantic similarity risks conflating textual proximity with evaluative agreement. We test whether embedding-based similarity reflects qualitative coding distinctions across rationale pairs in medical education. Similarity declined as rationale length differences grew and was less effective at distinguishing whether LLMs preserved the human’s central claim.
%U https://aclanthology.org/2026.aimecon-main.52/
%P 466-471
Markdown (Informal)
[Semantic Similarity is Not Enough for Comparing LLM and Human Rater Rationales](https://aclanthology.org/2026.aimecon-main.52/) (Jung et al., AIME-Con 2026)
ACL
- Julie Jongeun Jung, Max Lu, Dogus Darici, and Emilia Brüegge. 2026. Semantic Similarity is Not Enough for Comparing LLM and Human Rater Rationales. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers, pages 466–471, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).