@inproceedings{fae-etal-2026-comparison,
title = "A Comparison of Commonly Used Automatic Evaluation Metrics for Open-Ended Tasks in {P}ortuguese",
author = "Fa{\'e}, Eduardo D. and
Balreira, Dennis Giovani and
Moreira, Viviane Pereira",
editor = "Barbosa, Bryan Khelven da Silva and
Paes, Aline and
Felippo, Ariani Di",
booktitle = "Proceedings of the 17th {B}razilian Symposium in Information and Human Language Technology",
month = oct,
year = "2026",
address = "Cuiab{\'a}, Mato Grosso, Brazil",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.stil-1.11/",
doi = "10.5753/stil.2026.26578",
pages = "126--138",
abstract = "Open-ended tasks in Natural Language processing are characterized by the absence of a fixed set of outputs or short, predetermined responses. Typical examples include open-domain question answering and summarization. The automatic evaluation of these tasks is challenging, and existing metrics suffer from important limitations. Thus, many works have already been proposed to evaluate the performance of such metrics. However, the great majority of these works were conducted in English, leaving a gap for analyses in other languages, such as Portuguese. This work investigates the performance of some of the most popular automated intrinsic evaluation metrics for open-ended tasks by analyzing their correlation with human judgment. Using both the Summarization and Question Answering tasks, this study compares traditional n-gram-based metrics with metrics based on contextual embeddings generated by Pre-trained Language Models (PLMs), performing all analyses using models and datasets available for Portuguese. Results indicate that while PLM-based metrics generally show slightly higher correlation with human perception, their performance is highly reliant on the quality of the models used. Still, none of the evaluated metrics displayed a strong correlation with human judgments, suggesting the need for more robust metrics."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="fae-etal-2026-comparison">
<titleInfo>
<title>A Comparison of Commonly Used Automatic Evaluation Metrics for Open-Ended Tasks in Portuguese</title>
</titleInfo>
<name type="personal">
<namePart type="given">Eduardo</namePart>
<namePart type="given">D</namePart>
<namePart type="family">Faé</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dennis</namePart>
<namePart type="given">Giovani</namePart>
<namePart type="family">Balreira</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Viviane</namePart>
<namePart type="given">Pereira</namePart>
<namePart type="family">Moreira</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bryan</namePart>
<namePart type="given">Khelven</namePart>
<namePart type="given">da</namePart>
<namePart type="given">Silva</namePart>
<namePart type="family">Barbosa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ariani</namePart>
<namePart type="given">Di</namePart>
<namePart type="family">Felippo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Cuiabá, Mato Grosso, Brazil</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Open-ended tasks in Natural Language processing are characterized by the absence of a fixed set of outputs or short, predetermined responses. Typical examples include open-domain question answering and summarization. The automatic evaluation of these tasks is challenging, and existing metrics suffer from important limitations. Thus, many works have already been proposed to evaluate the performance of such metrics. However, the great majority of these works were conducted in English, leaving a gap for analyses in other languages, such as Portuguese. This work investigates the performance of some of the most popular automated intrinsic evaluation metrics for open-ended tasks by analyzing their correlation with human judgment. Using both the Summarization and Question Answering tasks, this study compares traditional n-gram-based metrics with metrics based on contextual embeddings generated by Pre-trained Language Models (PLMs), performing all analyses using models and datasets available for Portuguese. Results indicate that while PLM-based metrics generally show slightly higher correlation with human perception, their performance is highly reliant on the quality of the models used. Still, none of the evaluated metrics displayed a strong correlation with human judgments, suggesting the need for more robust metrics.</abstract>
<identifier type="citekey">fae-etal-2026-comparison</identifier>
<identifier type="doi">10.5753/stil.2026.26578</identifier>
<location>
<url>https://aclanthology.org/2026.stil-1.11/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>126</start>
<end>138</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T A Comparison of Commonly Used Automatic Evaluation Metrics for Open-Ended Tasks in Portuguese
%A Faé, Eduardo D.
%A Balreira, Dennis Giovani
%A Moreira, Viviane Pereira
%Y Barbosa, Bryan Khelven da Silva
%Y Paes, Aline
%Y Felippo, Ariani Di
%S Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology
%D 2026
%8 October
%I Association for Computational Linguistics
%C Cuiabá, Mato Grosso, Brazil
%F fae-etal-2026-comparison
%X Open-ended tasks in Natural Language processing are characterized by the absence of a fixed set of outputs or short, predetermined responses. Typical examples include open-domain question answering and summarization. The automatic evaluation of these tasks is challenging, and existing metrics suffer from important limitations. Thus, many works have already been proposed to evaluate the performance of such metrics. However, the great majority of these works were conducted in English, leaving a gap for analyses in other languages, such as Portuguese. This work investigates the performance of some of the most popular automated intrinsic evaluation metrics for open-ended tasks by analyzing their correlation with human judgment. Using both the Summarization and Question Answering tasks, this study compares traditional n-gram-based metrics with metrics based on contextual embeddings generated by Pre-trained Language Models (PLMs), performing all analyses using models and datasets available for Portuguese. Results indicate that while PLM-based metrics generally show slightly higher correlation with human perception, their performance is highly reliant on the quality of the models used. Still, none of the evaluated metrics displayed a strong correlation with human judgments, suggesting the need for more robust metrics.
%R 10.5753/stil.2026.26578
%U https://aclanthology.org/2026.stil-1.11/
%U https://doi.org/10.5753/stil.2026.26578
%P 126-138
Markdown (Informal)
[A Comparison of Commonly Used Automatic Evaluation Metrics for Open-Ended Tasks in Portuguese](https://aclanthology.org/2026.stil-1.11/) (Faé et al., STIL 2026)
ACL