@inproceedings{vintar-etal-2026-charting,
title = "Charting the {E}uropean {LLM} Benchmarking Landscape: A New Taxonomy and Registry",
author = "Vintar, Spela and
Brglez, Mojca and
Kuzman Punger{\v{s}}ek, Taja and
Ljube{\v{s}}i{\'c}, Nikola",
editor = "Montejo-Raez, Arturo and
Grisot, Cristina and
Blochowiak, Joanna and
Ljube{\v{s}}i{\'c}, Nikola and
Battaner, Elena and
Rigau, German",
booktitle = "Proceedings of Shaping Multilingual, Multimodal {AI} for the Social Sciences and Humanities ({LLM}s4{SSH}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma de Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.llms4ssh-1.22/",
doi = "10.63317/4kixe3c9zmde",
pages = "205--217",
abstract = "While new benchmarks for large language models (LLMs) are being developed continuously to catch up with the growing capabilities of new models and AI in general, using and evaluating LLMs in non-English languages remains a poorly-charted landscape. We give a concise overview of recent developments in LLM benchmarking, and then propose a new taxonomy for the categorization of benchmarks that is tailored to multilingual or non-English use scenarios. We further propose a registry of benchmarks implementing the new categorization and documenting benchmarks with a rich set of metadescriptors. While still at a pilot stage, such a registry can lead to a more coordinated development of benchmarks for European languages. We conclude with a review of current trends and advocate for a higher language and culture sensitivity of evaluation methods."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="vintar-etal-2026-charting">
<titleInfo>
<title>Charting the European LLM Benchmarking Landscape: A New Taxonomy and Registry</title>
</titleInfo>
<name type="personal">
<namePart type="given">Spela</namePart>
<namePart type="family">Vintar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mojca</namePart>
<namePart type="family">Brglez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Taja</namePart>
<namePart type="family">Kuzman Pungeršek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nikola</namePart>
<namePart type="family">Ljubešić</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Shaping Multilingual, Multimodal AI for the Social Sciences and Humanities (LLMs4SSH) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Arturo</namePart>
<namePart type="family">Montejo-Raez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Cristina</namePart>
<namePart type="family">Grisot</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joanna</namePart>
<namePart type="family">Blochowiak</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nikola</namePart>
<namePart type="family">Ljubešić</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Battaner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">German</namePart>
<namePart type="family">Rigau</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>While new benchmarks for large language models (LLMs) are being developed continuously to catch up with the growing capabilities of new models and AI in general, using and evaluating LLMs in non-English languages remains a poorly-charted landscape. We give a concise overview of recent developments in LLM benchmarking, and then propose a new taxonomy for the categorization of benchmarks that is tailored to multilingual or non-English use scenarios. We further propose a registry of benchmarks implementing the new categorization and documenting benchmarks with a rich set of metadescriptors. While still at a pilot stage, such a registry can lead to a more coordinated development of benchmarks for European languages. We conclude with a review of current trends and advocate for a higher language and culture sensitivity of evaluation methods.</abstract>
<identifier type="citekey">vintar-etal-2026-charting</identifier>
<identifier type="doi">10.63317/4kixe3c9zmde</identifier>
<location>
<url>https://aclanthology.org/2026.llms4ssh-1.22/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>205</start>
<end>217</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Charting the European LLM Benchmarking Landscape: A New Taxonomy and Registry
%A Vintar, Spela
%A Brglez, Mojca
%A Kuzman Pungeršek, Taja
%A Ljubešić, Nikola
%Y Montejo-Raez, Arturo
%Y Grisot, Cristina
%Y Blochowiak, Joanna
%Y Ljubešić, Nikola
%Y Battaner, Elena
%Y Rigau, German
%S Proceedings of Shaping Multilingual, Multimodal AI for the Social Sciences and Humanities (LLMs4SSH) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca (Spain)
%F vintar-etal-2026-charting
%X While new benchmarks for large language models (LLMs) are being developed continuously to catch up with the growing capabilities of new models and AI in general, using and evaluating LLMs in non-English languages remains a poorly-charted landscape. We give a concise overview of recent developments in LLM benchmarking, and then propose a new taxonomy for the categorization of benchmarks that is tailored to multilingual or non-English use scenarios. We further propose a registry of benchmarks implementing the new categorization and documenting benchmarks with a rich set of metadescriptors. While still at a pilot stage, such a registry can lead to a more coordinated development of benchmarks for European languages. We conclude with a review of current trends and advocate for a higher language and culture sensitivity of evaluation methods.
%R 10.63317/4kixe3c9zmde
%U https://aclanthology.org/2026.llms4ssh-1.22/
%U https://doi.org/10.63317/4kixe3c9zmde
%P 205-217
Markdown (Informal)
[Charting the European LLM Benchmarking Landscape: A New Taxonomy and Registry](https://aclanthology.org/2026.llms4ssh-1.22/) (Vintar et al., LLMs4SSH 2026)
ACL