@inproceedings{einarsson-etal-2026-icelandic,
title = "{I}celandic Math Eval: A Competitive Mathematics Benchmark for Large Language Models",
author = {Einarsson, Hafsteinn and
Haraldsson, J{\"o}kull Ari and
Derayat, {\'I}var Armin and
Lund, Sigr{\'u}n Helga and
Magn{\'u}sson, Benedikt Steinar},
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.26/",
doi = "10.63317/43xpv5rqyhr6",
pages = "393--406",
abstract = "We introduce Icelandic Math Eval, the first comprehensive benchmark for evaluating large language models (LLMs) on competitive mathematics problems in Icelandic. Our dataset comprises 1,027 problems from Icelandic mathematics competitions spanning from 1984 to 2025, covering algebra, geometry, number theory, and combinatorics across ten difficulty levels. We evaluate three state-of-the-art models, Claude Sonnet 4.5, Gemini 2.5 Pro, and GPT-5, using a dual evaluation methodology that tests both with and without multiple-choice options. Our results reveal several key findings: (1) models achieve 81-93{\%} overall accuracy, demonstrating substantial cross-lingual transfer of mathematical reasoning capabilities; (2) a dramatic 17.5 percentage point performance drop on problems containing images highlights persistent challenges in multimodal mathematical reasoning; (3) a 6.7 percentage point gap between evaluation modes suggests that multiple-choice formats may overestimate genuine reasoning capabilities; and (4) systematic performance degradation with increasing difficulty, dropping to 43{\%} on the most challenging problems. Using an LLM-as-judge evaluation approach, we provide detailed analysis across problem types, difficulty levels, and model capabilities. This work contributes to multilingual AI evaluation and demonstrates the importance of developing rigorous benchmarks for diverse languages to ensure comprehensive assessment of AI capabilities."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="einarsson-etal-2026-icelandic">
<titleInfo>
<title>Icelandic Math Eval: A Competitive Mathematics Benchmark for Large Language Models</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hafsteinn</namePart>
<namePart type="family">Einarsson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jökull</namePart>
<namePart type="given">Ari</namePart>
<namePart type="family">Haraldsson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ívar</namePart>
<namePart type="given">Armin</namePart>
<namePart type="family">Derayat</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sigrún</namePart>
<namePart type="given">Helga</namePart>
<namePart type="family">Lund</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Benedikt</namePart>
<namePart type="given">Steinar</namePart>
<namePart type="family">Magnússon</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We introduce Icelandic Math Eval, the first comprehensive benchmark for evaluating large language models (LLMs) on competitive mathematics problems in Icelandic. Our dataset comprises 1,027 problems from Icelandic mathematics competitions spanning from 1984 to 2025, covering algebra, geometry, number theory, and combinatorics across ten difficulty levels. We evaluate three state-of-the-art models, Claude Sonnet 4.5, Gemini 2.5 Pro, and GPT-5, using a dual evaluation methodology that tests both with and without multiple-choice options. Our results reveal several key findings: (1) models achieve 81-93% overall accuracy, demonstrating substantial cross-lingual transfer of mathematical reasoning capabilities; (2) a dramatic 17.5 percentage point performance drop on problems containing images highlights persistent challenges in multimodal mathematical reasoning; (3) a 6.7 percentage point gap between evaluation modes suggests that multiple-choice formats may overestimate genuine reasoning capabilities; and (4) systematic performance degradation with increasing difficulty, dropping to 43% on the most challenging problems. Using an LLM-as-judge evaluation approach, we provide detailed analysis across problem types, difficulty levels, and model capabilities. This work contributes to multilingual AI evaluation and demonstrates the importance of developing rigorous benchmarks for diverse languages to ensure comprehensive assessment of AI capabilities.</abstract>
<identifier type="citekey">einarsson-etal-2026-icelandic</identifier>
<identifier type="doi">10.63317/43xpv5rqyhr6</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.26/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>393</start>
<end>406</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Icelandic Math Eval: A Competitive Mathematics Benchmark for Large Language Models
%A Einarsson, Hafsteinn
%A Haraldsson, Jökull Ari
%A Derayat, Ívar Armin
%A Lund, Sigrún Helga
%A Magnússon, Benedikt Steinar
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F einarsson-etal-2026-icelandic
%X We introduce Icelandic Math Eval, the first comprehensive benchmark for evaluating large language models (LLMs) on competitive mathematics problems in Icelandic. Our dataset comprises 1,027 problems from Icelandic mathematics competitions spanning from 1984 to 2025, covering algebra, geometry, number theory, and combinatorics across ten difficulty levels. We evaluate three state-of-the-art models, Claude Sonnet 4.5, Gemini 2.5 Pro, and GPT-5, using a dual evaluation methodology that tests both with and without multiple-choice options. Our results reveal several key findings: (1) models achieve 81-93% overall accuracy, demonstrating substantial cross-lingual transfer of mathematical reasoning capabilities; (2) a dramatic 17.5 percentage point performance drop on problems containing images highlights persistent challenges in multimodal mathematical reasoning; (3) a 6.7 percentage point gap between evaluation modes suggests that multiple-choice formats may overestimate genuine reasoning capabilities; and (4) systematic performance degradation with increasing difficulty, dropping to 43% on the most challenging problems. Using an LLM-as-judge evaluation approach, we provide detailed analysis across problem types, difficulty levels, and model capabilities. This work contributes to multilingual AI evaluation and demonstrates the importance of developing rigorous benchmarks for diverse languages to ensure comprehensive assessment of AI capabilities.
%R 10.63317/43xpv5rqyhr6
%U https://aclanthology.org/2026.lrec-1.26/
%U https://doi.org/10.63317/43xpv5rqyhr6
%P 393-406
Markdown (Informal)
[Icelandic Math Eval: A Competitive Mathematics Benchmark for Large Language Models](https://aclanthology.org/2026.lrec-1.26/) (Einarsson et al., LREC 2026)
ACL