@inproceedings{frost-2026-beyond,
title = "Beyond the Grid: Auditing Code-Editing Search for Automated Essay Scoring",
author = "Frost, Julius",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.54/",
pages = "423--440",
ISBN = "979-8-9983004-1-7",
abstract = "Code-editing language-model agents can change training choices beyond the hyper- parameter grids tested in automated essay scoring. Comparing these approaches requires distinguishing gains from a broader search space from evidence of a better search procedure. We compare code-editing agents with grid-restricted search, including random search, in two studies on ASAP-AES with nominally matched 12-hour search budgets. Code-editing produced the configuration with the highest test quadratic-weighted kappa (QWK) point estimate in each primary comparison. However, random search over a grid built afterwards around the Study 2 agent{'}s backbone, input length and head rule recovered most of its gain over BERT. Requiring a minimum validation-score improvement to accept a trial also left the accepted configuration{'}s validation score below the highest recorded valid validation score in every run that accepted a trial under this rule. The primary comparisons use one search run per condition, and some test folds reused essays involved in configuration selection, so these results compare selected configurations without establishing search-procedure superiority. These findings motivate reporting the available search choices and both accepted and best-scoring trials, and evaluating search procedures through repeated runs on data independent of configuration selection."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="frost-2026-beyond">
<titleInfo>
<title>Beyond the Grid: Auditing Code-Editing Search for Automated Essay Scoring</title>
</titleInfo>
<name type="personal">
<namePart type="given">Julius</namePart>
<namePart type="family">Frost</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>Code-editing language-model agents can change training choices beyond the hyper- parameter grids tested in automated essay scoring. Comparing these approaches requires distinguishing gains from a broader search space from evidence of a better search procedure. We compare code-editing agents with grid-restricted search, including random search, in two studies on ASAP-AES with nominally matched 12-hour search budgets. Code-editing produced the configuration with the highest test quadratic-weighted kappa (QWK) point estimate in each primary comparison. However, random search over a grid built afterwards around the Study 2 agent’s backbone, input length and head rule recovered most of its gain over BERT. Requiring a minimum validation-score improvement to accept a trial also left the accepted configuration’s validation score below the highest recorded valid validation score in every run that accepted a trial under this rule. The primary comparisons use one search run per condition, and some test folds reused essays involved in configuration selection, so these results compare selected configurations without establishing search-procedure superiority. These findings motivate reporting the available search choices and both accepted and best-scoring trials, and evaluating search procedures through repeated runs on data independent of configuration selection.</abstract>
<identifier type="citekey">frost-2026-beyond</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.54/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>423</start>
<end>440</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Beyond the Grid: Auditing Code-Editing Search for Automated Essay Scoring
%A Frost, Julius
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F frost-2026-beyond
%X Code-editing language-model agents can change training choices beyond the hyper- parameter grids tested in automated essay scoring. Comparing these approaches requires distinguishing gains from a broader search space from evidence of a better search procedure. We compare code-editing agents with grid-restricted search, including random search, in two studies on ASAP-AES with nominally matched 12-hour search budgets. Code-editing produced the configuration with the highest test quadratic-weighted kappa (QWK) point estimate in each primary comparison. However, random search over a grid built afterwards around the Study 2 agent’s backbone, input length and head rule recovered most of its gain over BERT. Requiring a minimum validation-score improvement to accept a trial also left the accepted configuration’s validation score below the highest recorded valid validation score in every run that accepted a trial under this rule. The primary comparisons use one search run per condition, and some test folds reused essays involved in configuration selection, so these results compare selected configurations without establishing search-procedure superiority. These findings motivate reporting the available search choices and both accepted and best-scoring trials, and evaluating search procedures through repeated runs on data independent of configuration selection.
%U https://aclanthology.org/2026.aimecon-wip.54/
%P 423-440
Markdown (Informal)
[Beyond the Grid: Auditing Code-Editing Search for Automated Essay Scoring](https://aclanthology.org/2026.aimecon-wip.54/) (Frost, AIME-Con 2026)
ACL