@article{soler-etal-2024-impact,
title = "The Impact of Word Splitting on the Semantic Content of Contextualized Word Representations",
author = "Soler, Aina Gar{\'\i} and
Labeau, Matthieu and
Clavel, Chlo{\'e}",
journal = "Transactions of the Association for Computational Linguistics",
volume = "12",
year = "2024",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2024.tacl-1.17",
doi = "10.1162/tacl_a_00647",
pages = "299--320",
abstract = "When deriving contextualized word representations from language models, a decision needs to be made on how to obtain one for out-of-vocabulary (OOV) words that are segmented into subwords. What is the best way to represent these words with a single vector, and are these representations of worse quality than those of in-vocabulary words? We carry out an intrinsic evaluation of embeddings from different models on semantic similarity tasks involving OOV words. Our analysis reveals, among other interesting findings, that the quality of representations of words that are split is often, but not always, worse than that of the embeddings of known words. Their similarity values, however, must be interpreted with caution.",
}
<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="soler-etal-2024-impact">
<titleInfo>
<title>The Impact of Word Splitting on the Semantic Content of Contextualized Word Representations</title>
</titleInfo>
<name type="personal">
<namePart type="given">Aina</namePart>
<namePart type="given">Garí</namePart>
<namePart type="family">Soler</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Matthieu</namePart>
<namePart type="family">Labeau</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chloé</namePart>
<namePart type="family">Clavel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2024</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>When deriving contextualized word representations from language models, a decision needs to be made on how to obtain one for out-of-vocabulary (OOV) words that are segmented into subwords. What is the best way to represent these words with a single vector, and are these representations of worse quality than those of in-vocabulary words? We carry out an intrinsic evaluation of embeddings from different models on semantic similarity tasks involving OOV words. Our analysis reveals, among other interesting findings, that the quality of representations of words that are split is often, but not always, worse than that of the embeddings of known words. Their similarity values, however, must be interpreted with caution.</abstract>
<identifier type="citekey">soler-etal-2024-impact</identifier>
<identifier type="doi">10.1162/tacl_a_00647</identifier>
<location>
<url>https://aclanthology.org/2024.tacl-1.17</url>
</location>
<part>
<date>2024</date>
<detail type="volume"><number>12</number></detail>
<extent unit="page">
<start>299</start>
<end>320</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T The Impact of Word Splitting on the Semantic Content of Contextualized Word Representations
%A Soler, Aina Garí
%A Labeau, Matthieu
%A Clavel, Chloé
%J Transactions of the Association for Computational Linguistics
%D 2024
%V 12
%I MIT Press
%C Cambridge, MA
%F soler-etal-2024-impact
%X When deriving contextualized word representations from language models, a decision needs to be made on how to obtain one for out-of-vocabulary (OOV) words that are segmented into subwords. What is the best way to represent these words with a single vector, and are these representations of worse quality than those of in-vocabulary words? We carry out an intrinsic evaluation of embeddings from different models on semantic similarity tasks involving OOV words. Our analysis reveals, among other interesting findings, that the quality of representations of words that are split is often, but not always, worse than that of the embeddings of known words. Their similarity values, however, must be interpreted with caution.
%R 10.1162/tacl_a_00647
%U https://aclanthology.org/2024.tacl-1.17
%U https://doi.org/10.1162/tacl_a_00647
%P 299-320
Markdown (Informal)
[The Impact of Word Splitting on the Semantic Content of Contextualized Word Representations](https://aclanthology.org/2024.tacl-1.17) (Soler et al., TACL 2024)
ACL