@inproceedings{zhang-etal-2026-small,
title = "Can Small Language Models Teach Math Through Socratic Dialogue?",
author = "Zhang, Dizhi and
Li, Mengchen and
Wang, Zichen and
Zeng, Ziqiao and
Mehta, Shruti and
de Souza, Talita de Paula Cypriano and
Isotani, Seiji",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Full Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-main.67/",
pages = "599--605",
ISBN = "979-8-9983004-0-0",
abstract = "This study evaluates whether three locally deployed small language models (LLaMA-3.2-1B, Qwen-2.5-0.5B, and Gemma-3-1B) can implement Socratic tutoring in Grade 6{--}8 mathematics. Using a rubric-based protocol across 72 sessions, results show that all models struggled to sustain guided inquiry, with Gemma performing best overall."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="zhang-etal-2026-small">
<titleInfo>
<title>Can Small Language Models Teach Math Through Socratic Dialogue?</title>
</titleInfo>
<name type="personal">
<namePart type="given">Dizhi</namePart>
<namePart type="family">Zhang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mengchen</namePart>
<namePart type="family">Li</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Zichen</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ziqiao</namePart>
<namePart type="family">Zeng</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shruti</namePart>
<namePart type="family">Mehta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Talita</namePart>
<namePart type="given">de</namePart>
<namePart type="given">Paula</namePart>
<namePart type="given">Cypriano</namePart>
<namePart type="family">de Souza</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Seiji</namePart>
<namePart type="family">Isotani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-0-0</identifier>
</relatedItem>
<abstract>This study evaluates whether three locally deployed small language models (LLaMA-3.2-1B, Qwen-2.5-0.5B, and Gemma-3-1B) can implement Socratic tutoring in Grade 6–8 mathematics. Using a rubric-based protocol across 72 sessions, results show that all models struggled to sustain guided inquiry, with Gemma performing best overall.</abstract>
<identifier type="citekey">zhang-etal-2026-small</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-main.67/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>599</start>
<end>605</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Can Small Language Models Teach Math Through Socratic Dialogue?
%A Zhang, Dizhi
%A Li, Mengchen
%A Wang, Zichen
%A Zeng, Ziqiao
%A Mehta, Shruti
%A de Souza, Talita de Paula Cypriano
%A Isotani, Seiji
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-0-0
%F zhang-etal-2026-small
%X This study evaluates whether three locally deployed small language models (LLaMA-3.2-1B, Qwen-2.5-0.5B, and Gemma-3-1B) can implement Socratic tutoring in Grade 6–8 mathematics. Using a rubric-based protocol across 72 sessions, results show that all models struggled to sustain guided inquiry, with Gemma performing best overall.
%U https://aclanthology.org/2026.aimecon-main.67/
%P 599-605
Markdown (Informal)
[Can Small Language Models Teach Math Through Socratic Dialogue?](https://aclanthology.org/2026.aimecon-main.67/) (Zhang et al., AIME-Con 2026)
ACL
- Dizhi Zhang, Mengchen Li, Zichen Wang, Ziqiao Zeng, Shruti Mehta, Talita de Paula Cypriano de Souza, and Seiji Isotani. 2026. Can Small Language Models Teach Math Through Socratic Dialogue?. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Full Papers, pages 599–605, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).