@inproceedings{arslan-etal-2026-automated,
title = "Automated Evaluation of Mathematical Equivalence Between Personalized and Standard Word Problems",
author = "Arslan, Burcu and
Choi, Ikkyu and
Sparks, Jesse R. and
Gooch, Reginald M. and
Walkington, Candace and
Bernacki, Matthew L.",
editor = "Wilson, Joshua and
Ormerod, Christopher and
Beiting-Parrish, Magdalen",
booktitle = "Proceedings of the Artificial Intelligence in Measurement and Education Conference ({AIME}-Con): Coordinated Session Papers",
month = oct,
year = "2026",
address = "Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States",
publisher = "National Council on Measurement in Education (NCME)",
url = "https://aclanthology.org/2026.aimecon-sessions.32/",
pages = "291--305",
ISBN = "979-8-9983004-2-4",
abstract = "Generative AI enables real-time and scalable context personalization of mathematics word problems (MWPs) based on students' self-reported interests during assessment. However, a question arises: are personalized and standard MWPs mathematically equivalent? In this paper, we present two Natural Language Processing pipelines for evaluating mathematical equivalence between personalized and standard MWPs."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="arslan-etal-2026-automated">
<titleInfo>
<title>Automated Evaluation of Mathematical Equivalence Between Personalized and Standard Word Problems</title>
</titleInfo>
<name type="personal">
<namePart type="given">Burcu</namePart>
<namePart type="family">Arslan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ikkyu</namePart>
<namePart type="family">Choi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jesse</namePart>
<namePart type="given">R</namePart>
<namePart type="family">Sparks</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Reginald</namePart>
<namePart type="given">M</namePart>
<namePart type="family">Gooch</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Candace</namePart>
<namePart type="family">Walkington</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Matthew</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Bernacki</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Joshua</namePart>
<namePart type="family">Wilson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Ormerod</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Magdalen</namePart>
<namePart type="family">Beiting-Parrish</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>National Council on Measurement in Education (NCME)</publisher>
<place>
<placeTerm type="text">Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-9983004-2-4</identifier>
</relatedItem>
<abstract>Generative AI enables real-time and scalable context personalization of mathematics word problems (MWPs) based on students’ self-reported interests during assessment. However, a question arises: are personalized and standard MWPs mathematically equivalent? In this paper, we present two Natural Language Processing pipelines for evaluating mathematical equivalence between personalized and standard MWPs.</abstract>
<identifier type="citekey">arslan-etal-2026-automated</identifier>
<location>
<url>https://aclanthology.org/2026.aimecon-sessions.32/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>291</start>
<end>305</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Automated Evaluation of Mathematical Equivalence Between Personalized and Standard Word Problems
%A Arslan, Burcu
%A Choi, Ikkyu
%A Sparks, Jesse R.
%A Gooch, Reginald M.
%A Walkington, Candace
%A Bernacki, Matthew L.
%Y Wilson, Joshua
%Y Ormerod, Christopher
%Y Beiting-Parrish, Magdalen
%S Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers
%D 2026
%8 October
%I National Council on Measurement in Education (NCME)
%C Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States
%@ 979-8-9983004-2-4
%F arslan-etal-2026-automated
%X Generative AI enables real-time and scalable context personalization of mathematics word problems (MWPs) based on students’ self-reported interests during assessment. However, a question arises: are personalized and standard MWPs mathematically equivalent? In this paper, we present two Natural Language Processing pipelines for evaluating mathematical equivalence between personalized and standard MWPs.
%U https://aclanthology.org/2026.aimecon-sessions.32/
%P 291-305
Markdown (Informal)
[Automated Evaluation of Mathematical Equivalence Between Personalized and Standard Word Problems](https://aclanthology.org/2026.aimecon-sessions.32/) (Arslan et al., AIME-Con 2026)
ACL
- Burcu Arslan, Ikkyu Choi, Jesse R. Sparks, Reginald M. Gooch, Candace Walkington, and Matthew L. Bernacki. 2026. Automated Evaluation of Mathematical Equivalence Between Personalized and Standard Word Problems. In Proceedings of the Artificial Intelligence in Measurement and Education Conference (AIME-Con): Coordinated Session Papers, pages 291–305, Wyndham Grand Pittsburgh Downtown, Pittsburgh, Pennsylvania, United States. National Council on Measurement in Education (NCME).