@inproceedings{vuong-etal-2026-assessing,
title = "Assessing Small Language Models as Decimal-Arithmetic Tutors: A Measurement Framework",
author = "Vuong, Mai Que and
Ahmadli, Shahana and
Zhou, Michelle and
Kim, Minseok and
Mehta, Shruti and
de Souza, Talita de Paula Cypriano and
Isotani, Seiji",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Works in Progress",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-wip.51/",
pages = "397--404",
ISBN = "979-8-9983004-1-7",
abstract = "Small language models (SLMs) are increasingly proposed for educational use because they promise lower cost, offline deployment, and stronger data privacy. We report an exploratory evaluation of three sub-2B-parameter models on example-based decimal-arithmetic tutoring. Across structured interactions, all three produced fluent, confident output that masked unstable pedagogy and frequent mathematical errors even on elementary decimal addition and place-value tasks. Building on these observations, we describe an emerging interaction-based measurement framework intended to support more defensible readiness decisions about SLMs as math tutors."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="vuong-etal-2026-assessing">
<titleInfo>
<title>Assessing Small Language Models as Decimal-Arithmetic Tutors: A Measurement Framework</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mai</namePart>
<namePart type="given">Que</namePart>
<namePart type="family">Vuong</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shahana</namePart>
<namePart type="family">Ahmadli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Michelle</namePart>
<namePart type="family">Zhou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Minseok</namePart>
<namePart type="family">Kim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shruti</namePart>
<namePart type="family">Mehta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Talita</namePart>
<namePart type="given">de</namePart>
<namePart type="given">Paula</namePart>
<namePart type="given">Cypriano</namePart>
<namePart type="family">de Souza</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Seiji</namePart>
<namePart type="family">Isotani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-1-7</identifier>
</relatedItem>
<abstract>Small language models (SLMs) are increasingly proposed for educational use because they promise lower cost, offline deployment, and stronger data privacy. We report an exploratory evaluation of three sub-2B-parameter models on example-based decimal-arithmetic tutoring. Across structured interactions, all three produced fluent, confident output that masked unstable pedagogy and frequent mathematical errors even on elementary decimal addition and place-value tasks. Building on these observations, we describe an emerging interaction-based measurement framework intended to support more defensible readiness decisions about SLMs as math tutors.</abstract>
<identifier type="citekey">vuong-etal-2026-assessing</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-wip.51/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>397</start>
<end>404</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Assessing Small Language Models as Decimal-Arithmetic Tutors: A Measurement Framework
%A Vuong, Mai Que
%A Ahmadli, Shahana
%A Zhou, Michelle
%A Kim, Minseok
%A Mehta, Shruti
%A de Souza, Talita de Paula Cypriano
%A Isotani, Seiji
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-1-7
%F vuong-etal-2026-assessing
%X Small language models (SLMs) are increasingly proposed for educational use because they promise lower cost, offline deployment, and stronger data privacy. We report an exploratory evaluation of three sub-2B-parameter models on example-based decimal-arithmetic tutoring. Across structured interactions, all three produced fluent, confident output that masked unstable pedagogy and frequent mathematical errors even on elementary decimal addition and place-value tasks. Building on these observations, we describe an emerging interaction-based measurement framework intended to support more defensible readiness decisions about SLMs as math tutors.
%U https://aclanthology.org/2026.aimecon-wip.51/
%P 397-404
Markdown (Informal)
[Assessing Small Language Models as Decimal-Arithmetic Tutors: A Measurement Framework](https://aclanthology.org/2026.aimecon-wip.51/) (Vuong et al., AIME-Con 2026)
ACL
- Mai Que Vuong, Shahana Ahmadli, Michelle Zhou, Minseok Kim, Shruti Mehta, Talita de Paula Cypriano de Souza, and Seiji Isotani. 2026. Assessing Small Language Models as Decimal-Arithmetic Tutors: A Measurement Framework. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Works in Progress, pages 397–404, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).