@inproceedings{miyaji-correa-2026-tail,
title = "Tail Smoothing and Cross-Lingual Volatility: Evaluating Estimative Uncertainty in Large Language Models for {B}razilian {P}ortuguese",
author = "Miyaji, Renato O. and
Corr{\^e}a, Pedro L. P.",
editor = "Barbosa, Bryan Khelven da Silva and
Paes, Aline and
Felippo, Ariani Di",
booktitle = "Proceedings of the 17th {B}razilian Symposium in Information and Human Language Technology",
month = oct,
year = "2026",
address = "Cuiab{\'a}, Mato Grosso, Brazil",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.stil-1.22/",
doi = "10.5753/stil.2026.24563",
pages = "258--269",
abstract = "This study evaluates how LLMs interpret Words of Estimative Probability (WEPs) in Brazilian Portuguese compared to English. We translated an English benchmark and compared multilingual models (GPT-5.1, Gemini 3 Flash) against a region-specific model (Sabi{\'a} 4). Our findings reveal a ``tail smoothing'' phenomenon, where models systematically compress extreme probabilities. Notably, while Gemini 3 Flash demonstrated remarkable cross-lingual stability, GPT-5.1 exhibited significant calibration degradation. Counterintuitively, when measured against the English human baseline, Sabi{\'a} 4 displayed the most aggressive distribution compression. These results suggest a complex dynamic: either linguistic fine-tuning does not ensure alignment, or it actively captures a culturally specific pragmatic ambiguity in Brazilian Portuguese. This exposes the vulnerabilities and nuances of deploying LLMs in nuanced semantic tasks in Brazilian Portuguese."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="miyaji-correa-2026-tail">
<titleInfo>
<title>Tail Smoothing and Cross-Lingual Volatility: Evaluating Estimative Uncertainty in Large Language Models for Brazilian Portuguese</title>
</titleInfo>
<name type="personal">
<namePart type="given">Renato</namePart>
<namePart type="given">O</namePart>
<namePart type="family">Miyaji</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pedro</namePart>
<namePart type="given">L</namePart>
<namePart type="given">P</namePart>
<namePart type="family">Corrêa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bryan</namePart>
<namePart type="given">Khelven</namePart>
<namePart type="given">da</namePart>
<namePart type="given">Silva</namePart>
<namePart type="family">Barbosa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ariani</namePart>
<namePart type="given">Di</namePart>
<namePart type="family">Felippo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Cuiabá, Mato Grosso, Brazil</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This study evaluates how LLMs interpret Words of Estimative Probability (WEPs) in Brazilian Portuguese compared to English. We translated an English benchmark and compared multilingual models (GPT-5.1, Gemini 3 Flash) against a region-specific model (Sabiá 4). Our findings reveal a “tail smoothing” phenomenon, where models systematically compress extreme probabilities. Notably, while Gemini 3 Flash demonstrated remarkable cross-lingual stability, GPT-5.1 exhibited significant calibration degradation. Counterintuitively, when measured against the English human baseline, Sabiá 4 displayed the most aggressive distribution compression. These results suggest a complex dynamic: either linguistic fine-tuning does not ensure alignment, or it actively captures a culturally specific pragmatic ambiguity in Brazilian Portuguese. This exposes the vulnerabilities and nuances of deploying LLMs in nuanced semantic tasks in Brazilian Portuguese.</abstract>
<identifier type="citekey">miyaji-correa-2026-tail</identifier>
<identifier type="doi">10.5753/stil.2026.24563</identifier>
<location>
<url>https://aclanthology.org/2026.stil-1.22/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>258</start>
<end>269</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Tail Smoothing and Cross-Lingual Volatility: Evaluating Estimative Uncertainty in Large Language Models for Brazilian Portuguese
%A Miyaji, Renato O.
%A Corrêa, Pedro L. P.
%Y Barbosa, Bryan Khelven da Silva
%Y Paes, Aline
%Y Felippo, Ariani Di
%S Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology
%D 2026
%8 October
%I Association for Computational Linguistics
%C Cuiabá, Mato Grosso, Brazil
%F miyaji-correa-2026-tail
%X This study evaluates how LLMs interpret Words of Estimative Probability (WEPs) in Brazilian Portuguese compared to English. We translated an English benchmark and compared multilingual models (GPT-5.1, Gemini 3 Flash) against a region-specific model (Sabiá 4). Our findings reveal a “tail smoothing” phenomenon, where models systematically compress extreme probabilities. Notably, while Gemini 3 Flash demonstrated remarkable cross-lingual stability, GPT-5.1 exhibited significant calibration degradation. Counterintuitively, when measured against the English human baseline, Sabiá 4 displayed the most aggressive distribution compression. These results suggest a complex dynamic: either linguistic fine-tuning does not ensure alignment, or it actively captures a culturally specific pragmatic ambiguity in Brazilian Portuguese. This exposes the vulnerabilities and nuances of deploying LLMs in nuanced semantic tasks in Brazilian Portuguese.
%R 10.5753/stil.2026.24563
%U https://aclanthology.org/2026.stil-1.22/
%U https://doi.org/10.5753/stil.2026.24563
%P 258-269
Markdown (Informal)
[Tail Smoothing and Cross-Lingual Volatility: Evaluating Estimative Uncertainty in Large Language Models for Brazilian Portuguese](https://aclanthology.org/2026.stil-1.22/) (Miyaji & Corrêa, STIL 2026)
ACL