@inproceedings{beigman-klebanov-etal-2026-towards,
title = "Towards evaluating teacher performance in a {G}en{AI} teaching simulation of a science discussion",
author = "Beigman Klebanov, Beata and
Mikeska, Jamie N. and
Zhao, Mengxuan and
Flynn, Catherine and
Fetrow, Devon and
Halder, Shreyashi and
Ubale, Rutuja and
Maxwell, Tricia and
Suhan, Michael",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.64/",
pages = "571--580",
ISBN = "979-8-9983004-0-0",
abstract = "GenAI can power simulated student agents that provide opportunities for educators to engage in core teaching practices, such as leading a small group argumentation-based science discussion. To realize the potential of such simulations and support teacher reflection and learning, it is necessary to provide participants with timely feedback on their performance in the simulation. This study investigates systems for automated evaluation of and feedback on teacher performance in a simulation along the dimension of making use of student ideas to move the discussion forward. We address three research questions: (a) How well do models fine-tuned on transcripts of teacher performance in a matching human-puppeteered teaching simulation (that served as the model during the development of the GenAI one) perform in evaluating transcripts from the GenAI teaching simulation? (b) How well does a system using a few-shot LLM perform on the same task? (c) How do educators perceive the quality and usefulness of the automatically generated feedback? The findings underscore the importance of a rigorous evaluation of automated evaluation and feedback systems."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="beigman-klebanov-etal-2026-towards">
<titleInfo>
<title>Towards evaluating teacher performance in a GenAI teaching simulation of a science discussion</title>
</titleInfo>
<name type="personal">
<namePart type="given">Beata</namePart>
<namePart type="family">Beigman Klebanov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jamie</namePart>
<namePart type="given">N</namePart>
<namePart type="family">Mikeska</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mengxuan</namePart>
<namePart type="family">Zhao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Catherine</namePart>
<namePart type="family">Flynn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Devon</namePart>
<namePart type="family">Fetrow</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shreyashi</namePart>
<namePart type="family">Halder</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rutuja</namePart>
<namePart type="family">Ubale</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tricia</namePart>
<namePart type="family">Maxwell</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Michael</namePart>
<namePart type="family">Suhan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>GenAI can power simulated student agents that provide opportunities for educators to engage in core teaching practices, such as leading a small group argumentation-based science discussion. To realize the potential of such simulations and support teacher reflection and learning, it is necessary to provide participants with timely feedback on their performance in the simulation. This study investigates systems for automated evaluation of and feedback on teacher performance in a simulation along the dimension of making use of student ideas to move the discussion forward. We address three research questions: (a) How well do models fine-tuned on transcripts of teacher performance in a matching human-puppeteered teaching simulation (that served as the model during the development of the GenAI one) perform in evaluating transcripts from the GenAI teaching simulation? (b) How well does a system using a few-shot LLM perform on the same task? (c) How do educators perceive the quality and usefulness of the automatically generated feedback? The findings underscore the importance of a rigorous evaluation of automated evaluation and feedback systems.</abstract>
<identifier type="citekey">beigman-klebanov-etal-2026-towards</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.64/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>571</start>
<end>580</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Towards evaluating teacher performance in a GenAI teaching simulation of a science discussion
%A Beigman Klebanov, Beata
%A Mikeska, Jamie N.
%A Zhao, Mengxuan
%A Flynn, Catherine
%A Fetrow, Devon
%A Halder, Shreyashi
%A Ubale, Rutuja
%A Maxwell, Tricia
%A Suhan, Michael
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F beigman-klebanov-etal-2026-towards
%X GenAI can power simulated student agents that provide opportunities for educators to engage in core teaching practices, such as leading a small group argumentation-based science discussion. To realize the potential of such simulations and support teacher reflection and learning, it is necessary to provide participants with timely feedback on their performance in the simulation. This study investigates systems for automated evaluation of and feedback on teacher performance in a simulation along the dimension of making use of student ideas to move the discussion forward. We address three research questions: (a) How well do models fine-tuned on transcripts of teacher performance in a matching human-puppeteered teaching simulation (that served as the model during the development of the GenAI one) perform in evaluating transcripts from the GenAI teaching simulation? (b) How well does a system using a few-shot LLM perform on the same task? (c) How do educators perceive the quality and usefulness of the automatically generated feedback? The findings underscore the importance of a rigorous evaluation of automated evaluation and feedback systems.
%U https://aclanthology.org/2026.aimecon-main.64/
%P 571-580
Markdown (Informal)
[Towards evaluating teacher performance in a GenAI teaching simulation of a science discussion](https://aclanthology.org/2026.aimecon-main.64/) (Beigman Klebanov et al., AIME-Con 2026)
ACL
- Beata Beigman Klebanov, Jamie N. Mikeska, Mengxuan Zhao, Catherine Flynn, Devon Fetrow, Shreyashi Halder, Rutuja Ubale, Tricia Maxwell, and Michael Suhan. 2026. Towards evaluating teacher performance in a GenAI teaching simulation of a science discussion. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers, pages 571–580, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).