@article{haslett-etal-2026-byte,
title = "Byte-pair Encoding Captures Aspects of Word Meaning that Distributional Semantics Overlooks",
author = "Haslett, David A. and
Chan, Antoni B. and
Hsiao, Janet H.",
journal = "Transactions of the Association for Computational Linguistics",
volume = "14",
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.tacl-1.113/",
doi = "10.1162/tacl.a.810",
pages = "2425--2448",
abstract = "Large language models learn word meanings through distributional semantics (i.e., patterns in text), and distribution does not always capture information about taxonomy and function, which are essential to human representations of word meaning. However, sound and spelling often conveys such information, and language models decompose many words into subword tokens, which may identify category markers. For example, GPT-5 segments fluorine and bromine into fluor + ine and brom + ine, which share the category marker ine. We provide evidence that, across 18 languages, shared tokens predict greater semantic priming effects in humans (e.g., people recognize bromine faster after reading fluorine), beyond what distributional similarity explains. Furthermore, across seven languages, shared tokens boost semantic priming beyond what shared morphemes explain. This suggests that subword tokens could help language models arrive at human-like representations of word meanings. However, we also provide evidence that the trend towards larger vocabularies of tokens obscures category markers, and even when language models have access to subword tokens, they underestimate the semantic relatedness of words that share tokens."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="haslett-etal-2026-byte">
<titleInfo>
<title>Byte-pair Encoding Captures Aspects of Word Meaning that Distributional Semantics Overlooks</title>
</titleInfo>
<name type="personal">
<namePart type="given">David</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Haslett</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antoni</namePart>
<namePart type="given">B</namePart>
<namePart type="family">Chan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Janet</namePart>
<namePart type="given">H</namePart>
<namePart type="family">Hsiao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>Large language models learn word meanings through distributional semantics (i.e., patterns in text), and distribution does not always capture information about taxonomy and function, which are essential to human representations of word meaning. However, sound and spelling often conveys such information, and language models decompose many words into subword tokens, which may identify category markers. For example, GPT-5 segments fluorine and bromine into fluor + ine and brom + ine, which share the category marker ine. We provide evidence that, across 18 languages, shared tokens predict greater semantic priming effects in humans (e.g., people recognize bromine faster after reading fluorine), beyond what distributional similarity explains. Furthermore, across seven languages, shared tokens boost semantic priming beyond what shared morphemes explain. This suggests that subword tokens could help language models arrive at human-like representations of word meanings. However, we also provide evidence that the trend towards larger vocabularies of tokens obscures category markers, and even when language models have access to subword tokens, they underestimate the semantic relatedness of words that share tokens.</abstract>
<identifier type="citekey">haslett-etal-2026-byte</identifier>
<identifier type="doi">10.1162/tacl.a.810</identifier>
<location>
<url>https://aclanthology.org/2026.tacl-1.113/</url>
</location>
<part>
<date>2026</date>
<detail type="volume"><number>14</number></detail>
<extent unit="page">
<start>2425</start>
<end>2448</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T Byte-pair Encoding Captures Aspects of Word Meaning that Distributional Semantics Overlooks
%A Haslett, David A.
%A Chan, Antoni B.
%A Hsiao, Janet H.
%J Transactions of the Association for Computational Linguistics
%D 2026
%V 14
%I MIT Press
%C Cambridge, MA
%F haslett-etal-2026-byte
%X Large language models learn word meanings through distributional semantics (i.e., patterns in text), and distribution does not always capture information about taxonomy and function, which are essential to human representations of word meaning. However, sound and spelling often conveys such information, and language models decompose many words into subword tokens, which may identify category markers. For example, GPT-5 segments fluorine and bromine into fluor + ine and brom + ine, which share the category marker ine. We provide evidence that, across 18 languages, shared tokens predict greater semantic priming effects in humans (e.g., people recognize bromine faster after reading fluorine), beyond what distributional similarity explains. Furthermore, across seven languages, shared tokens boost semantic priming beyond what shared morphemes explain. This suggests that subword tokens could help language models arrive at human-like representations of word meanings. However, we also provide evidence that the trend towards larger vocabularies of tokens obscures category markers, and even when language models have access to subword tokens, they underestimate the semantic relatedness of words that share tokens.
%R 10.1162/tacl.a.810
%U https://aclanthology.org/2026.tacl-1.113/
%U https://doi.org/10.1162/tacl.a.810
%P 2425-2448
Markdown (Informal)
[Byte-pair Encoding Captures Aspects of Word Meaning that Distributional Semantics Overlooks](https://aclanthology.org/2026.tacl-1.113/) (Haslett et al., TACL 2026)
ACL