@inproceedings{hochgraf-etal-2026-beyond,
title = "Beyond Aggregate Scores: A Diagnostic Analysis of {P}ortuguese {LLM} Evaluation Suites",
author = "Hochgraf, Gustavo and
Assis, Gabriel and
Paes, Aline and
Cozman, Fabio G.",
editor = "Barbosa, Bryan Khelven da Silva and
Paes, Aline and
Felippo, Ariani Di",
booktitle = "Proceedings of the 17th {B}razilian Symposium in Information and Human Language Technology",
month = oct,
year = "2026",
address = "Cuiab{\'a}, Mato Grosso, Brazil",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.stil-1.18/",
doi = "10.5753/stil.2026.26539",
pages = "209--217",
abstract = "Benchmarks play a central role in evaluating large language models (LLMs), providing standardized comparisons across models, tasks, and adaptation strategies. However, aggregate benchmark scores often compress performance into a single number, obscuring important variation across tasks. This issue is especially relevant for less-resourced languages like Portuguese, where benchmarks may combine native tasks with translated datasets spanning heterogeneous categories and subareas. In this work, we examine this issue using PoETa v2, a broad Portuguese evaluation benchmark, as a diagnostic setting to evaluate three Qwen3 1.7B variants under different Portuguese adaptation settings. Our results show that, although overall scores differ by less than one percentage point across models, disaggregated analyses reveal substantially different patterns across task origin, category, and subarea. In particular, gains on native Portuguese tasks may coexist with losses on translated tasks, while different adaptation corpora redistribute performance in distinct ways. Taken together, our findings highlight the importance of complementing aggregate scores with multi-level analyses when evaluating Portuguese LLMs, particularly in the context of language-specific adaptation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hochgraf-etal-2026-beyond">
<titleInfo>
<title>Beyond Aggregate Scores: A Diagnostic Analysis of Portuguese LLM Evaluation Suites</title>
</titleInfo>
<name type="personal">
<namePart type="given">Gustavo</namePart>
<namePart type="family">Hochgraf</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gabriel</namePart>
<namePart type="family">Assis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fabio</namePart>
<namePart type="given">G</namePart>
<namePart type="family">Cozman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bryan</namePart>
<namePart type="given">Khelven</namePart>
<namePart type="given">da</namePart>
<namePart type="given">Silva</namePart>
<namePart type="family">Barbosa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ariani</namePart>
<namePart type="given">Di</namePart>
<namePart type="family">Felippo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Cuiabá, Mato Grosso, Brazil</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Benchmarks play a central role in evaluating large language models (LLMs), providing standardized comparisons across models, tasks, and adaptation strategies. However, aggregate benchmark scores often compress performance into a single number, obscuring important variation across tasks. This issue is especially relevant for less-resourced languages like Portuguese, where benchmarks may combine native tasks with translated datasets spanning heterogeneous categories and subareas. In this work, we examine this issue using PoETa v2, a broad Portuguese evaluation benchmark, as a diagnostic setting to evaluate three Qwen3 1.7B variants under different Portuguese adaptation settings. Our results show that, although overall scores differ by less than one percentage point across models, disaggregated analyses reveal substantially different patterns across task origin, category, and subarea. In particular, gains on native Portuguese tasks may coexist with losses on translated tasks, while different adaptation corpora redistribute performance in distinct ways. Taken together, our findings highlight the importance of complementing aggregate scores with multi-level analyses when evaluating Portuguese LLMs, particularly in the context of language-specific adaptation.</abstract>
<identifier type="citekey">hochgraf-etal-2026-beyond</identifier>
<identifier type="doi">10.5753/stil.2026.26539</identifier>
<location>
<url>https://aclanthology.org/2026.stil-1.18/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>209</start>
<end>217</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Beyond Aggregate Scores: A Diagnostic Analysis of Portuguese LLM Evaluation Suites
%A Hochgraf, Gustavo
%A Assis, Gabriel
%A Paes, Aline
%A Cozman, Fabio G.
%Y Barbosa, Bryan Khelven da Silva
%Y Paes, Aline
%Y Felippo, Ariani Di
%S Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology
%D 2026
%8 October
%I Association for Computational Linguistics
%C Cuiabá, Mato Grosso, Brazil
%F hochgraf-etal-2026-beyond
%X Benchmarks play a central role in evaluating large language models (LLMs), providing standardized comparisons across models, tasks, and adaptation strategies. However, aggregate benchmark scores often compress performance into a single number, obscuring important variation across tasks. This issue is especially relevant for less-resourced languages like Portuguese, where benchmarks may combine native tasks with translated datasets spanning heterogeneous categories and subareas. In this work, we examine this issue using PoETa v2, a broad Portuguese evaluation benchmark, as a diagnostic setting to evaluate three Qwen3 1.7B variants under different Portuguese adaptation settings. Our results show that, although overall scores differ by less than one percentage point across models, disaggregated analyses reveal substantially different patterns across task origin, category, and subarea. In particular, gains on native Portuguese tasks may coexist with losses on translated tasks, while different adaptation corpora redistribute performance in distinct ways. Taken together, our findings highlight the importance of complementing aggregate scores with multi-level analyses when evaluating Portuguese LLMs, particularly in the context of language-specific adaptation.
%R 10.5753/stil.2026.26539
%U https://aclanthology.org/2026.stil-1.18/
%U https://doi.org/10.5753/stil.2026.26539
%P 209-217
Markdown (Informal)
[Beyond Aggregate Scores: A Diagnostic Analysis of Portuguese LLM Evaluation Suites](https://aclanthology.org/2026.stil-1.18/) (Hochgraf et al., STIL 2026)
ACL