@inproceedings{lee-etal-2026-llm-assessor,
title = "{LLM}-As-An-Assessor: Can Open-Weight {LLM}s Assess Computational Thinking via Student-Designed Embodied Games?",
author = "Lee, William and
Gattupalli, Sai and
Arroyo, Ivon and
O{'}Connor, Brendan",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.37/",
pages = "289--297",
ISBN = "979-8-9983004-1-7",
abstract = "We evaluate whether open-weight LLMs can assess Computational Thinking in middle-school students' finite-state game designs against human labels. Across a rich and a sparse design, models detect some behaviors at near-human agreement but over-credit absent ones and fail at counting, loop detection, and standards mapping. Failures trace to fixable setup."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="lee-etal-2026-llm-assessor">
<titleInfo>
<title>LLM-As-An-Assessor: Can Open-Weight LLMs Assess Computational Thinking via Student-Designed Embodied Games?</title>
</titleInfo>
<name type="personal">
<namePart type="given">William</namePart>
<namePart type="family">Lee</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sai</namePart>
<namePart type="family">Gattupalli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ivon</namePart>
<namePart type="family">Arroyo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Brendan</namePart>
<namePart type="family">O’Connor</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>We evaluate whether open-weight LLMs can assess Computational Thinking in middle-school students’ finite-state game designs against human labels. Across a rich and a sparse design, models detect some behaviors at near-human agreement but over-credit absent ones and fail at counting, loop detection, and standards mapping. Failures trace to fixable setup.</abstract>
<identifier type="citekey">lee-etal-2026-llm-assessor</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.37/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>289</start>
<end>297</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T LLM-As-An-Assessor: Can Open-Weight LLMs Assess Computational Thinking via Student-Designed Embodied Games?
%A Lee, William
%A Gattupalli, Sai
%A Arroyo, Ivon
%A O’Connor, Brendan
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F lee-etal-2026-llm-assessor
%X We evaluate whether open-weight LLMs can assess Computational Thinking in middle-school students’ finite-state game designs against human labels. Across a rich and a sparse design, models detect some behaviors at near-human agreement but over-credit absent ones and fail at counting, loop detection, and standards mapping. Failures trace to fixable setup.
%U https://aclanthology.org/2026.aimecon-wip.37/
%P 289-297
Markdown (Informal)
[LLM-As-An-Assessor: Can Open-Weight LLMs Assess Computational Thinking via Student-Designed Embodied Games?](https://aclanthology.org/2026.aimecon-wip.37/) (Lee et al., AIME-Con 2026)
ACL