@inproceedings{herrera-etal-2026-exploration,
title = "Exploration of Sentence Representations in {S}panish {BERT}-like Models",
author = "Herrera, Gonzalo and
Ros{\'a}, Aiala and
Chiruzzo, Luis",
editor = "Claramunt, German Rigau and
Gamallo, Pablo and
Mu{\~n}oz Guillena, Rafael and
Chiruzzo, Luis and
Mart{\'i}nez C{\'a}mara, Eugenio",
booktitle = "Proceedings of {LANLP}: Bridging {I}bero and {L}atin {A}merican {NLP} Communities",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.lanlp-1.8/",
doi = "10.63317/5moein87oxiw",
pages = "55--65",
abstract = "Transformer-based language models, ubiquitous in NLP nowadays, generate internal representations (embeddings) of words and sentences. Yet, systematic comparisons of embedding strategies from various models remain limited. In this work, we evaluate Spanish embeddings from several BERT-like models (BETO, multilingual BERT, XLM-RoBERTa, ROUBERTa) to understand their syntactic and semantic capabilities across layers. We propose novel sentence-level analogy tests to probe generalization. Results show tasks like verb negation or word reordering perform best with embeddings from earlier layers, while nuanced semantic distinctions{---}such as agent or patient gender{---}are better captured by deeper layers. Our findings provide guidelines for embedding strategies and offer a foundation for further NLP research."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="herrera-etal-2026-exploration">
<titleInfo>
<title>Exploration of Sentence Representations in Spanish BERT-like Models</title>
</titleInfo>
<name type="personal">
<namePart type="given">Gonzalo</namePart>
<namePart type="family">Herrera</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aiala</namePart>
<namePart type="family">Rosá</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Luis</namePart>
<namePart type="family">Chiruzzo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of LANLP: Bridging Ibero and Latin American NLP Communities</title>
</titleInfo>
<name type="personal">
<namePart type="given">German</namePart>
<namePart type="given">Rigau</namePart>
<namePart type="family">Claramunt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pablo</namePart>
<namePart type="family">Gamallo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rafael</namePart>
<namePart type="family">Muñoz Guillena</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Luis</namePart>
<namePart type="family">Chiruzzo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eugenio</namePart>
<namePart type="family">Martínez Cámara</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Transformer-based language models, ubiquitous in NLP nowadays, generate internal representations (embeddings) of words and sentences. Yet, systematic comparisons of embedding strategies from various models remain limited. In this work, we evaluate Spanish embeddings from several BERT-like models (BETO, multilingual BERT, XLM-RoBERTa, ROUBERTa) to understand their syntactic and semantic capabilities across layers. We propose novel sentence-level analogy tests to probe generalization. Results show tasks like verb negation or word reordering perform best with embeddings from earlier layers, while nuanced semantic distinctions—such as agent or patient gender—are better captured by deeper layers. Our findings provide guidelines for embedding strategies and offer a foundation for further NLP research.</abstract>
<identifier type="citekey">herrera-etal-2026-exploration</identifier>
<identifier type="doi">10.63317/5moein87oxiw</identifier>
<location>
<url>https://aclanthology.org/2026.lanlp-1.8/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>55</start>
<end>65</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Exploration of Sentence Representations in Spanish BERT-like Models
%A Herrera, Gonzalo
%A Rosá, Aiala
%A Chiruzzo, Luis
%Y Claramunt, German Rigau
%Y Gamallo, Pablo
%Y Muñoz Guillena, Rafael
%Y Chiruzzo, Luis
%Y Martínez Cámara, Eugenio
%S Proceedings of LANLP: Bridging Ibero and Latin American NLP Communities
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F herrera-etal-2026-exploration
%X Transformer-based language models, ubiquitous in NLP nowadays, generate internal representations (embeddings) of words and sentences. Yet, systematic comparisons of embedding strategies from various models remain limited. In this work, we evaluate Spanish embeddings from several BERT-like models (BETO, multilingual BERT, XLM-RoBERTa, ROUBERTa) to understand their syntactic and semantic capabilities across layers. We propose novel sentence-level analogy tests to probe generalization. Results show tasks like verb negation or word reordering perform best with embeddings from earlier layers, while nuanced semantic distinctions—such as agent or patient gender—are better captured by deeper layers. Our findings provide guidelines for embedding strategies and offer a foundation for further NLP research.
%R 10.63317/5moein87oxiw
%U https://aclanthology.org/2026.lanlp-1.8/
%U https://doi.org/10.63317/5moein87oxiw
%P 55-65
Markdown (Informal)
[Exploration of Sentence Representations in Spanish BERT-like Models](https://aclanthology.org/2026.lanlp-1.8/) (Herrera et al., LANLP 2026)
ACL