@inproceedings{assis-etal-2026-llm,
title = "Do {LLM} Judges Agree with Humans? Evaluating Financial Commentaries from Material Facts",
author = "Assis, Gabriel and
Vianna, Daniela and
Real, Livy and
Masid, Marina Ramalhete and
Nepomuceno, Jo{\~a}o and
Rottschaefer, Eduardo and
da Silva, Altigran Soares and
Paes, Aline",
editor = "Barbosa, Bryan Khelven da Silva and
Paes, Aline and
Felippo, Ariani Di",
booktitle = "Proceedings of the 17th {B}razilian Symposium in Information and Human Language Technology",
month = oct,
year = "2026",
address = "Cuiab{\'a}, Mato Grosso, Brazil",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.stil-1.3/",
doi = "10.5753/stil.2026.26484",
pages = "26--40",
abstract = "The rapid adoption of large language models (LLMs) in finance has enabled automated generation of in-domain texts such as earnings summaries, market analyses, and commentary on regulated disclosures. Generating accurate and accessible financial commentary from material facts poses challenges related to domain-specific language, strict factual faithfulness, and readability. Evaluating such outputs is difficult: traditional automatic metrics overlook financial correctness and adequacy, while human expert assessment is reliable but costly and subjective. This paper proposes a multidimensional evaluation protocol for generating financial commentary in Portuguese and investigates the use of LLMs as evaluators ({``}LLMs-as-judges'') in this high-stakes, low-resource setting. We systematically compare human expert judgments and LLM-based evaluation preferences, while also investigating complementary dimensions such as writing quality, factuality, usefulness, and simplicity. To the best of our knowledge, this is the first proposal of an evaluation protocol for this task in Portuguese. Furthermore, we analyze where LLM judgments align with or diverge from human assessments, providing practical insights and recommendations for evaluation methodologies in financial AI applications."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="assis-etal-2026-llm">
<titleInfo>
<title>Do LLM Judges Agree with Humans? Evaluating Financial Commentaries from Material Facts</title>
</titleInfo>
<name type="personal">
<namePart type="given">Gabriel</namePart>
<namePart type="family">Assis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Daniela</namePart>
<namePart type="family">Vianna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Livy</namePart>
<namePart type="family">Real</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marina</namePart>
<namePart type="given">Ramalhete</namePart>
<namePart type="family">Masid</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">João</namePart>
<namePart type="family">Nepomuceno</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eduardo</namePart>
<namePart type="family">Rottschaefer</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Altigran</namePart>
<namePart type="given">Soares</namePart>
<namePart type="family">da Silva</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bryan</namePart>
<namePart type="given">Khelven</namePart>
<namePart type="given">da</namePart>
<namePart type="given">Silva</namePart>
<namePart type="family">Barbosa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ariani</namePart>
<namePart type="given">Di</namePart>
<namePart type="family">Felippo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Cuiabá, Mato Grosso, Brazil</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The rapid adoption of large language models (LLMs) in finance has enabled automated generation of in-domain texts such as earnings summaries, market analyses, and commentary on regulated disclosures. Generating accurate and accessible financial commentary from material facts poses challenges related to domain-specific language, strict factual faithfulness, and readability. Evaluating such outputs is difficult: traditional automatic metrics overlook financial correctness and adequacy, while human expert assessment is reliable but costly and subjective. This paper proposes a multidimensional evaluation protocol for generating financial commentary in Portuguese and investigates the use of LLMs as evaluators (“LLMs-as-judges”) in this high-stakes, low-resource setting. We systematically compare human expert judgments and LLM-based evaluation preferences, while also investigating complementary dimensions such as writing quality, factuality, usefulness, and simplicity. To the best of our knowledge, this is the first proposal of an evaluation protocol for this task in Portuguese. Furthermore, we analyze where LLM judgments align with or diverge from human assessments, providing practical insights and recommendations for evaluation methodologies in financial AI applications.</abstract>
<identifier type="citekey">assis-etal-2026-llm</identifier>
<identifier type="doi">10.5753/stil.2026.26484</identifier>
<location>
<url>https://aclanthology.org/2026.stil-1.3/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>26</start>
<end>40</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Do LLM Judges Agree with Humans? Evaluating Financial Commentaries from Material Facts
%A Assis, Gabriel
%A Vianna, Daniela
%A Real, Livy
%A Masid, Marina Ramalhete
%A Nepomuceno, João
%A Rottschaefer, Eduardo
%A da Silva, Altigran Soares
%A Paes, Aline
%Y Barbosa, Bryan Khelven da Silva
%Y Paes, Aline
%Y Felippo, Ariani Di
%S Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology
%D 2026
%8 October
%I Association for Computational Linguistics
%C Cuiabá, Mato Grosso, Brazil
%F assis-etal-2026-llm
%X The rapid adoption of large language models (LLMs) in finance has enabled automated generation of in-domain texts such as earnings summaries, market analyses, and commentary on regulated disclosures. Generating accurate and accessible financial commentary from material facts poses challenges related to domain-specific language, strict factual faithfulness, and readability. Evaluating such outputs is difficult: traditional automatic metrics overlook financial correctness and adequacy, while human expert assessment is reliable but costly and subjective. This paper proposes a multidimensional evaluation protocol for generating financial commentary in Portuguese and investigates the use of LLMs as evaluators (“LLMs-as-judges”) in this high-stakes, low-resource setting. We systematically compare human expert judgments and LLM-based evaluation preferences, while also investigating complementary dimensions such as writing quality, factuality, usefulness, and simplicity. To the best of our knowledge, this is the first proposal of an evaluation protocol for this task in Portuguese. Furthermore, we analyze where LLM judgments align with or diverge from human assessments, providing practical insights and recommendations for evaluation methodologies in financial AI applications.
%R 10.5753/stil.2026.26484
%U https://aclanthology.org/2026.stil-1.3/
%U https://doi.org/10.5753/stil.2026.26484
%P 26-40
Markdown (Informal)
[Do LLM Judges Agree with Humans? Evaluating Financial Commentaries from Material Facts](https://aclanthology.org/2026.stil-1.3/) (Assis et al., STIL 2026)
ACL
- Gabriel Assis, Daniela Vianna, Livy Real, Marina Ramalhete Masid, João Nepomuceno, Eduardo Rottschaefer, Altigran Soares da Silva, and Aline Paes. 2026. Do LLM Judges Agree with Humans? Evaluating Financial Commentaries from Material Facts. In Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology, pages 26–40, Cuiabá, Mato Grosso, Brazil. Association for Computational Linguistics.