@inproceedings{hemenway-bellows-2026-construct,
title = "Construct Validity of Small-Sample Transformer Scoring Models: A Mechanistic Interpretability Approach",
author = "Hemenway, Michael P. and
Bellows, Martha",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.41/",
pages = "322--328",
ISBN = "979-8-9983004-1-7",
abstract = "Small-sample transformer scoring models ( n = 64 n=64) reach high human agreement but risk leaning on surface shortcuts like response length. Evaluating Mechanistic Interpretability strategies across 90 models, we show correlational methods suffer from seed noise, whereas interventional erasure proves small-sample models causally depend more on length. We outline an actionable audit protocol."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hemenway-bellows-2026-construct">
<titleInfo>
<title>Construct Validity of Small-Sample Transformer Scoring Models: A Mechanistic Interpretability Approach</title>
</titleInfo>
<name type="personal">
<namePart type="given">Michael</namePart>
<namePart type="given">P</namePart>
<namePart type="family">Hemenway</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Martha</namePart>
<namePart type="family">Bellows</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>Small-sample transformer scoring models ( n = 64 n=64) reach high human agreement but risk leaning on surface shortcuts like response length. Evaluating Mechanistic Interpretability strategies across 90 models, we show correlational methods suffer from seed noise, whereas interventional erasure proves small-sample models causally depend more on length. We outline an actionable audit protocol.</abstract>
<identifier type="citekey">hemenway-bellows-2026-construct</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.41/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>322</start>
<end>328</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Construct Validity of Small-Sample Transformer Scoring Models: A Mechanistic Interpretability Approach
%A Hemenway, Michael P.
%A Bellows, Martha
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F hemenway-bellows-2026-construct
%X Small-sample transformer scoring models ( n = 64 n=64) reach high human agreement but risk leaning on surface shortcuts like response length. Evaluating Mechanistic Interpretability strategies across 90 models, we show correlational methods suffer from seed noise, whereas interventional erasure proves small-sample models causally depend more on length. We outline an actionable audit protocol.
%U https://aclanthology.org/2026.aimecon-wip.41/
%P 322-328
Markdown (Informal)
[Construct Validity of Small-Sample Transformer Scoring Models: A Mechanistic Interpretability Approach](https://aclanthology.org/2026.aimecon-wip.41/) (Hemenway & Bellows, AIME-Con 2026)
ACL