@inproceedings{xiao-etal-2026-developing,
title = "Developing an {LLM} Tutor Quality Evaluation Scale",
author = "Xiao, Michael and
Tian, Zewei and
Liu, Alex and
Esbenshade, Lief and
Sun, Min",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Coordinated Session Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-sessions.15/",
pages = "124--145",
ISBN = "979-8-9983004-2-4",
abstract = "We introduce a unified metric set for evaluating LLM tutors by inductively coding 177 recent literature-based metrics and scoring tutoring dialogues with three LLM judges. Exploratory factor analysis identified five constructs and a strong general factor. We present the resulting 10-category, 41-item framework for confirmatory analysis and human validation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="xiao-etal-2026-developing">
<titleInfo>
<title>Developing an LLM Tutor Quality Evaluation Scale</title>
</titleInfo>
<name type="personal">
<namePart type="given">Michael</namePart>
<namePart type="family">Xiao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Zewei</namePart>
<namePart type="family">Tian</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alex</namePart>
<namePart type="family">Liu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lief</namePart>
<namePart type="family">Esbenshade</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Min</namePart>
<namePart type="family">Sun</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-2-4</identifier>
</relatedItem>
<abstract>We introduce a unified metric set for evaluating LLM tutors by inductively coding 177 recent literature-based metrics and scoring tutoring dialogues with three LLM judges. Exploratory factor analysis identified five constructs and a strong general factor. We present the resulting 10-category, 41-item framework for confirmatory analysis and human validation.</abstract>
<identifier type="citekey">xiao-etal-2026-developing</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-sessions.15/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>124</start>
<end>145</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Developing an LLM Tutor Quality Evaluation Scale
%A Xiao, Michael
%A Tian, Zewei
%A Liu, Alex
%A Esbenshade, Lief
%A Sun, Min
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-2-4
%F xiao-etal-2026-developing
%X We introduce a unified metric set for evaluating LLM tutors by inductively coding 177 recent literature-based metrics and scoring tutoring dialogues with three LLM judges. Exploratory factor analysis identified five constructs and a strong general factor. We present the resulting 10-category, 41-item framework for confirmatory analysis and human validation.
%U https://aclanthology.org/2026.aimecon-sessions.15/
%P 124-145
Markdown (Informal)
[Developing an LLM Tutor Quality Evaluation Scale](https://aclanthology.org/2026.aimecon-sessions.15/) (Xiao et al., AIME-Con 2026)
ACL
- Michael Xiao, Zewei Tian, Alex Liu, Lief Esbenshade, and Min Sun. 2026. Developing an LLM Tutor Quality Evaluation Scale. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers, pages 124–145, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).