@inproceedings{balter-etal-2026-multiplication,
title = "Multiplication in Multimodal {LLM}s: Computation with Text, Image, and Audio Inputs",
author = "Balter, Samuel Gideon and
Jerzak, Ethan and
Jerzak, Connor Thomas",
editor = "Liakata, Maria and
Moreira, Viviane P. and
Zhang, Jiajun and
Jurgens, David",
booktitle = "Findings of the {A}ssociation for {C}omputational {L}inguistics: {ACL} 2026",
month = jul,
year = "2026",
address = "San Diego, California, United States",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.findings-acl.2025/",
doi = "10.18653/v1/2026.findings-acl.2025",
pages = "40766--40780",
ISBN = "979-8-89176-395-1",
abstract = "Multimodal LLMs can accurately perceive numerical content across modalities yet fail to perform exact multi-digit multiplication when the identical underlying arithmetic problem is presented as numerals, number words, images, or in audio form. Because existing benchmarks often lack systematically paired instances across modalities, it remains difficult to compare genuine arithmetic limits within and across model families. We therefore introduce a controlled multimodal multiplication benchmark that factorially varies digit length, digit sparsity, representation (e.g., numerals vs. number words), and modality (text, rendered images, and audio), with paired instances from a reproducible generator. We also define arithmetic load, $C$, as the product of the total and non-zero digit number as a compact, mechanistically motivated proxy for operation count. Across evaluations, accuracy falls sharply as $C$ grows, often nearing zero by $C > 100$. Indeed, $C$ remains predictive of performance across modalities and models, with $R^2 > 0.5$, nearing the value from more complex measures of arithmetic load that count the number of intermediate arithmetic steps. A separate perception-versus-computation decomposition shows that multimodal degradation is primarily computational rather than perceptual: on matched perception checks, models are near-perfect ($>99\%$) across modalities even when multiplication accuracy drops substantially. Beyond measuring when models fail, we ask which procedures they are predisposed to follow. We introduce a style-controlled forced-completion loss probe that scores heuristic-specific reasoning prefixes{---}including columnar multiplication, distributive decomposition, and rounding/compensation. Here, distributive decomposition is favored in both text and vision modalities; heuristic-specific LoRA adapters produce near-orthogonal updates yet degrade accuracy, indicating the base model maintains a well-tuned internal router."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="balter-etal-2026-multiplication">
<titleInfo>
<title>Multiplication in Multimodal LLMs: Computation with Text, Image, and Audio Inputs</title>
</titleInfo>
<name type="personal">
<namePart type="given">Samuel</namePart>
<namePart type="given">Gideon</namePart>
<namePart type="family">Balter</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ethan</namePart>
<namePart type="family">Jerzak</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Connor</namePart>
<namePart type="given">Thomas</namePart>
<namePart type="family">Jerzak</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-07</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Findings of the Association for Computational Linguistics: ACL 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="family">Liakata</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Viviane</namePart>
<namePart type="given">P</namePart>
<namePart type="family">Moreira</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jiajun</namePart>
<namePart type="family">Zhang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">David</namePart>
<namePart type="family">Jurgens</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">San Diego, California, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-89176-395-1</identifier>
</relatedItem>
<abstract>Multimodal LLMs can accurately perceive numerical content across modalities yet fail to perform exact multi-digit multiplication when the identical underlying arithmetic problem is presented as numerals, number words, images, or in audio form. Because existing benchmarks often lack systematically paired instances across modalities, it remains difficult to compare genuine arithmetic limits within and across model families. We therefore introduce a controlled multimodal multiplication benchmark that factorially varies digit length, digit sparsity, representation (e.g., numerals vs. number words), and modality (text, rendered images, and audio), with paired instances from a reproducible generator. We also define arithmetic load, C, as the product of the total and non-zero digit number as a compact, mechanistically motivated proxy for operation count. Across evaluations, accuracy falls sharply as C grows, often nearing zero by C > 100. Indeed, C remains predictive of performance across modalities and models, with R² > 0.5, nearing the value from more complex measures of arithmetic load that count the number of intermediate arithmetic steps. A separate perception-versus-computation decomposition shows that multimodal degradation is primarily computational rather than perceptual: on matched perception checks, models are near-perfect (>99%) across modalities even when multiplication accuracy drops substantially. Beyond measuring when models fail, we ask which procedures they are predisposed to follow. We introduce a style-controlled forced-completion loss probe that scores heuristic-specific reasoning prefixes—including columnar multiplication, distributive decomposition, and rounding/compensation. Here, distributive decomposition is favored in both text and vision modalities; heuristic-specific LoRA adapters produce near-orthogonal updates yet degrade accuracy, indicating the base model maintains a well-tuned internal router.</abstract>
<identifier type="citekey">balter-etal-2026-multiplication</identifier>
<identifier type="doi">10.18653/v1/2026.findings-acl.2025</identifier>
<location>
<url>https://aclanthology.org/2026.findings-acl.2025/</url>
</location>
<part>
<date>2026-07</date>
<extent unit="page">
<start>40766</start>
<end>40780</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Multiplication in Multimodal LLMs: Computation with Text, Image, and Audio Inputs
%A Balter, Samuel Gideon
%A Jerzak, Ethan
%A Jerzak, Connor Thomas
%Y Liakata, Maria
%Y Moreira, Viviane P.
%Y Zhang, Jiajun
%Y Jurgens, David
%S Findings of the Association for Computational Linguistics: ACL 2026
%D 2026
%8 July
%I Association for Computational Linguistics
%C San Diego, California, United States
%@ 979-8-89176-395-1
%F balter-etal-2026-multiplication
%X Multimodal LLMs can accurately perceive numerical content across modalities yet fail to perform exact multi-digit multiplication when the identical underlying arithmetic problem is presented as numerals, number words, images, or in audio form. Because existing benchmarks often lack systematically paired instances across modalities, it remains difficult to compare genuine arithmetic limits within and across model families. We therefore introduce a controlled multimodal multiplication benchmark that factorially varies digit length, digit sparsity, representation (e.g., numerals vs. number words), and modality (text, rendered images, and audio), with paired instances from a reproducible generator. We also define arithmetic load, C, as the product of the total and non-zero digit number as a compact, mechanistically motivated proxy for operation count. Across evaluations, accuracy falls sharply as C grows, often nearing zero by C > 100. Indeed, C remains predictive of performance across modalities and models, with R² > 0.5, nearing the value from more complex measures of arithmetic load that count the number of intermediate arithmetic steps. A separate perception-versus-computation decomposition shows that multimodal degradation is primarily computational rather than perceptual: on matched perception checks, models are near-perfect (>99%) across modalities even when multiplication accuracy drops substantially. Beyond measuring when models fail, we ask which procedures they are predisposed to follow. We introduce a style-controlled forced-completion loss probe that scores heuristic-specific reasoning prefixes—including columnar multiplication, distributive decomposition, and rounding/compensation. Here, distributive decomposition is favored in both text and vision modalities; heuristic-specific LoRA adapters produce near-orthogonal updates yet degrade accuracy, indicating the base model maintains a well-tuned internal router.
%R 10.18653/v1/2026.findings-acl.2025
%U https://aclanthology.org/2026.findings-acl.2025/
%U https://doi.org/10.18653/v1/2026.findings-acl.2025
%P 40766-40780
Markdown (Informal)
[Multiplication in Multimodal LLMs: Computation with Text, Image, and Audio Inputs](https://aclanthology.org/2026.findings-acl.2025/) (Balter et al., Findings 2026)
ACL