@inproceedings{mello-finger-2026-evaluating,
title = "Evaluating {P}ortuguese Tokenizers as Morpheme Sequence in Relation to {LLM} Downstream Performance",
author = "Mello, Guilherme L. and
Finger, Marcelo",
editor = "Barbosa, Bryan Khelven da Silva and
Paes, Aline and
Felippo, Ariani Di",
booktitle = "Proceedings of the 17th {B}razilian Symposium in Information and Human Language Technology",
month = oct,
year = "2026",
address = "Cuiab{\'a}, Mato Grosso, Brazil",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.stil-1.21/",
doi = "10.5753/stil.2026.26562",
pages = "245--257",
abstract = "Sub-word tokenizers allow us to handle open vocabulary problems using a relatively small set of tokens. Despite its ease of use, it usually relies on a data-driven approach that does not directly employ linguistic or morphologic features for text tokenization. [Bostrom and Durrett 2020] and [Hofmann et al. 2021] demonstrate that morphemes improve LLM performance on English texts. In this work, we explore the hypothesis that tokenizers that are capable of producing a token sequence better aligned with a morpheme sequence can improve LLMs performance on Brazilian Portuguese. To evaluate how the presence of morphemes can impact LLMs for Brazilian Portuguese, we propose MorphEval-PT, a new evaluation procedure based on the psycholinguistic concept of morphological models of word processing. We build new BPE and Unigram vocabularies that are evaluated on MorphEval-PT and, in order to validate the impact of morphemes on LLMs performance, train new LLMs from scratch and evaluate its performance on downstream tasks. Consistently, BPE demonstrates a higher precision score than Unigram in its ability to represent morphemes, as well as better performance on every downstream task. These promising results indicate that accessing the ability of tokenizers to represent morphemes is an important feature in the development of LLMs for Brazilian Portuguese and that MorphEval-PT is a good and lightweight method to improve LLM performance before any pre-training."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="mello-finger-2026-evaluating">
<titleInfo>
<title>Evaluating Portuguese Tokenizers as Morpheme Sequence in Relation to LLM Downstream Performance</title>
</titleInfo>
<name type="personal">
<namePart type="given">Guilherme</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Mello</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marcelo</namePart>
<namePart type="family">Finger</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bryan</namePart>
<namePart type="given">Khelven</namePart>
<namePart type="given">da</namePart>
<namePart type="given">Silva</namePart>
<namePart type="family">Barbosa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ariani</namePart>
<namePart type="given">Di</namePart>
<namePart type="family">Felippo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Cuiabá, Mato Grosso, Brazil</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Sub-word tokenizers allow us to handle open vocabulary problems using a relatively small set of tokens. Despite its ease of use, it usually relies on a data-driven approach that does not directly employ linguistic or morphologic features for text tokenization. [Bostrom and Durrett 2020] and [Hofmann et al. 2021] demonstrate that morphemes improve LLM performance on English texts. In this work, we explore the hypothesis that tokenizers that are capable of producing a token sequence better aligned with a morpheme sequence can improve LLMs performance on Brazilian Portuguese. To evaluate how the presence of morphemes can impact LLMs for Brazilian Portuguese, we propose MorphEval-PT, a new evaluation procedure based on the psycholinguistic concept of morphological models of word processing. We build new BPE and Unigram vocabularies that are evaluated on MorphEval-PT and, in order to validate the impact of morphemes on LLMs performance, train new LLMs from scratch and evaluate its performance on downstream tasks. Consistently, BPE demonstrates a higher precision score than Unigram in its ability to represent morphemes, as well as better performance on every downstream task. These promising results indicate that accessing the ability of tokenizers to represent morphemes is an important feature in the development of LLMs for Brazilian Portuguese and that MorphEval-PT is a good and lightweight method to improve LLM performance before any pre-training.</abstract>
<identifier type="citekey">mello-finger-2026-evaluating</identifier>
<identifier type="doi">10.5753/stil.2026.26562</identifier>
<location>
<url>https://aclanthology.org/2026.stil-1.21/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>245</start>
<end>257</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Evaluating Portuguese Tokenizers as Morpheme Sequence in Relation to LLM Downstream Performance
%A Mello, Guilherme L.
%A Finger, Marcelo
%Y Barbosa, Bryan Khelven da Silva
%Y Paes, Aline
%Y Felippo, Ariani Di
%S Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology
%D 2026
%8 October
%I Association for Computational Linguistics
%C Cuiabá, Mato Grosso, Brazil
%F mello-finger-2026-evaluating
%X Sub-word tokenizers allow us to handle open vocabulary problems using a relatively small set of tokens. Despite its ease of use, it usually relies on a data-driven approach that does not directly employ linguistic or morphologic features for text tokenization. [Bostrom and Durrett 2020] and [Hofmann et al. 2021] demonstrate that morphemes improve LLM performance on English texts. In this work, we explore the hypothesis that tokenizers that are capable of producing a token sequence better aligned with a morpheme sequence can improve LLMs performance on Brazilian Portuguese. To evaluate how the presence of morphemes can impact LLMs for Brazilian Portuguese, we propose MorphEval-PT, a new evaluation procedure based on the psycholinguistic concept of morphological models of word processing. We build new BPE and Unigram vocabularies that are evaluated on MorphEval-PT and, in order to validate the impact of morphemes on LLMs performance, train new LLMs from scratch and evaluate its performance on downstream tasks. Consistently, BPE demonstrates a higher precision score than Unigram in its ability to represent morphemes, as well as better performance on every downstream task. These promising results indicate that accessing the ability of tokenizers to represent morphemes is an important feature in the development of LLMs for Brazilian Portuguese and that MorphEval-PT is a good and lightweight method to improve LLM performance before any pre-training.
%R 10.5753/stil.2026.26562
%U https://aclanthology.org/2026.stil-1.21/
%U https://doi.org/10.5753/stil.2026.26562
%P 245-257
Markdown (Informal)
[Evaluating Portuguese Tokenizers as Morpheme Sequence in Relation to LLM Downstream Performance](https://aclanthology.org/2026.stil-1.21/) (Mello & Finger, STIL 2026)
ACL