@inproceedings{xu-etal-2026-effects,
title = "Effects of Demographic Representations and Model Training Methods on Automated Scoring Engines",
author = "Xu, Yangmeng and
Bellows, Martha and
Wolfe, Edward W",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.16/",
pages = "154--160",
ISBN = "979-8-9983004-0-0",
abstract = "This study evaluated whether automated scoring engines maintain stable performance when training data composition and training methods vary. We manipulated demographic representation (gender, English language learner, race, student with disabilities) and compared feature-based versus transformer-based models. Results showed performances were stable across subgroup-representation densities and transformer models exhibited greater stability."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="xu-etal-2026-effects">
<titleInfo>
<title>Effects of Demographic Representations and Model Training Methods on Automated Scoring Engines</title>
</titleInfo>
<name type="personal">
<namePart type="given">Yangmeng</namePart>
<namePart type="family">Xu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Martha</namePart>
<namePart type="family">Bellows</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Edward</namePart>
<namePart type="given">W</namePart>
<namePart type="family">Wolfe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>This study evaluated whether automated scoring engines maintain stable performance when training data composition and training methods vary. We manipulated demographic representation (gender, English language learner, race, student with disabilities) and compared feature-based versus transformer-based models. Results showed performances were stable across subgroup-representation densities and transformer models exhibited greater stability.</abstract>
<identifier type="citekey">xu-etal-2026-effects</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.16/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>154</start>
<end>160</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Effects of Demographic Representations and Model Training Methods on Automated Scoring Engines
%A Xu, Yangmeng
%A Bellows, Martha
%A Wolfe, Edward W.
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F xu-etal-2026-effects
%X This study evaluated whether automated scoring engines maintain stable performance when training data composition and training methods vary. We manipulated demographic representation (gender, English language learner, race, student with disabilities) and compared feature-based versus transformer-based models. Results showed performances were stable across subgroup-representation densities and transformer models exhibited greater stability.
%U https://aclanthology.org/2026.aimecon-main.16/
%P 154-160
Markdown (Informal)
[Effects of Demographic Representations and Model Training Methods on Automated Scoring Engines](https://aclanthology.org/2026.aimecon-main.16/) (Xu et al., AIME-Con 2026)
ACL