@inproceedings{jacobs-etal-2026-much,
title = "How Much Does Hyperparameter Tuning Actually Help? An Efficiency Survey for Fine-Tuning Transformers to Score Mathematics-Explanation Items",
author = "Jacobs, Gregory M. and
Bediwy, Ahmed H. and
Bellows, Martha",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.26/",
pages = "200--206",
ISBN = "979-8-9983004-1-7",
abstract = "Fine-tuning pre-trained transformer models for constructed-response items often begins with a hyperparameter grid search using k-fold cross-validation. We study how much that search actually helps by fine-tuning two encoders{---}MathBERT, a smaller math-focused model, and DeBERTa-v3-large, a larger general-purpose model{---}to score 18 math-explanation items from a state-wide assessment. Crossing learning rate, weight decay, and label smoothing over two epochs and five folds (1,800 model-fold runs), we compare each model{'}s tuning gain directly against fold-to-fold noise via a gain-to-noise ratio. Gains were small relative to fold noise for both encoders: MathBERT{'}s ratio fell below one (0.90), and DeBERTa{'}s nominally higher ratio (1.42) traced to a handful of divergent fits rather than an informative search landscape. Weight decay and label smoothing were effectively inert, leaving learning rate as the only hyperparameter worth checking{---}though even for learning rate the best configurations offered small performance gains over a reasonable default. We accordingly recommend a lean workflow that fixes the inert hyperparameters to sensible defaults, runs a narrow learning-rate search extended modestly upward, and increases the epoch budget with early stopping, substantially reducing training compute without sacrificing accuracy."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="jacobs-etal-2026-much">
<titleInfo>
<title>How Much Does Hyperparameter Tuning Actually Help? An Efficiency Survey for Fine-Tuning Transformers to Score Mathematics-Explanation Items</title>
</titleInfo>
<name type="personal">
<namePart type="given">Gregory</namePart>
<namePart type="given">M</namePart>
<namePart type="family">Jacobs</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ahmed</namePart>
<namePart type="given">H</namePart>
<namePart type="family">Bediwy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Martha</namePart>
<namePart type="family">Bellows</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>Fine-tuning pre-trained transformer models for constructed-response items often begins with a hyperparameter grid search using k-fold cross-validation. We study how much that search actually helps by fine-tuning two encoders—MathBERT, a smaller math-focused model, and DeBERTa-v3-large, a larger general-purpose model—to score 18 math-explanation items from a state-wide assessment. Crossing learning rate, weight decay, and label smoothing over two epochs and five folds (1,800 model-fold runs), we compare each model’s tuning gain directly against fold-to-fold noise via a gain-to-noise ratio. Gains were small relative to fold noise for both encoders: MathBERT’s ratio fell below one (0.90), and DeBERTa’s nominally higher ratio (1.42) traced to a handful of divergent fits rather than an informative search landscape. Weight decay and label smoothing were effectively inert, leaving learning rate as the only hyperparameter worth checking—though even for learning rate the best configurations offered small performance gains over a reasonable default. We accordingly recommend a lean workflow that fixes the inert hyperparameters to sensible defaults, runs a narrow learning-rate search extended modestly upward, and increases the epoch budget with early stopping, substantially reducing training compute without sacrificing accuracy.</abstract>
<identifier type="citekey">jacobs-etal-2026-much</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.26/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>200</start>
<end>206</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T How Much Does Hyperparameter Tuning Actually Help? An Efficiency Survey for Fine-Tuning Transformers to Score Mathematics-Explanation Items
%A Jacobs, Gregory M.
%A Bediwy, Ahmed H.
%A Bellows, Martha
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F jacobs-etal-2026-much
%X Fine-tuning pre-trained transformer models for constructed-response items often begins with a hyperparameter grid search using k-fold cross-validation. We study how much that search actually helps by fine-tuning two encoders—MathBERT, a smaller math-focused model, and DeBERTa-v3-large, a larger general-purpose model—to score 18 math-explanation items from a state-wide assessment. Crossing learning rate, weight decay, and label smoothing over two epochs and five folds (1,800 model-fold runs), we compare each model’s tuning gain directly against fold-to-fold noise via a gain-to-noise ratio. Gains were small relative to fold noise for both encoders: MathBERT’s ratio fell below one (0.90), and DeBERTa’s nominally higher ratio (1.42) traced to a handful of divergent fits rather than an informative search landscape. Weight decay and label smoothing were effectively inert, leaving learning rate as the only hyperparameter worth checking—though even for learning rate the best configurations offered small performance gains over a reasonable default. We accordingly recommend a lean workflow that fixes the inert hyperparameters to sensible defaults, runs a narrow learning-rate search extended modestly upward, and increases the epoch budget with early stopping, substantially reducing training compute without sacrificing accuracy.
%U https://aclanthology.org/2026.aimecon-wip.26/
%P 200-206
Markdown (Informal)
[How Much Does Hyperparameter Tuning Actually Help? An Efficiency Survey for Fine-Tuning Transformers to Score Mathematics-Explanation Items](https://aclanthology.org/2026.aimecon-wip.26/) (Jacobs et al., AIME-Con 2026)
ACL