@inproceedings{brans-bloem-2026-multi,
title = "Multi-{S}im{L}ex for {D}utch: Benchmarking Embedding- and Prompt-Based Model Performance on Semantic Similarity",
author = "Brans, Lizzy and
Bloem, Jelke",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.380/",
doi = "10.63317/2q9dcx9cvnu9",
pages = "4846--4860",
abstract = "We introduce Dutch Multi-SimLex, a 1,888{--}pair extension of the Multi-SimLex benchmark for evaluating lexical semantic similarity in Dutch. The dataset was rated by 100 native speakers on a 0{--}6 scale and shows high reliability (overall ICC(2,k)=0.82) as well as strong alignment with English ({\ensuremath{\rho}}=0.73). Using this resource, we evaluate eighteen models across four architectural families: static embeddings, encoder-only transformers, encoder{--}decoders, and decoder-only LLMs. We evaluate models using two complementary approaches: embedding-based cosine similarity and prompted similarity judgments in Dutch. In embedding-based evaluation, FastText ({\ensuremath{\rho}}=0.485) and the monolingual Dutch encoder BERTje ({\ensuremath{\rho}}=0.468) achieve the strongest alignment with human ratings, while multilingual encoders such as mBERT ({\ensuremath{\rho}}=0.208) and XLM-R ({\ensuremath{\rho}}=0.186) perform weaker. Prompt-based evaluation yields substantially higher correlations, with GPT-4 ({\ensuremath{\rho}}=0.761) performing best, followed by DeepSeek-V3 ({\ensuremath{\rho}}=0.753) and Gemini 1.5 Pro ({\ensuremath{\rho}}=0.722). Together, the results show that model performance depends strongly on how meaning is tested. Dutch Multi-SimLex provides a reliable foundation for evaluating meaning across architectures and advancing Dutch semantic evaluation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="brans-bloem-2026-multi">
<titleInfo>
<title>Multi-SimLex for Dutch: Benchmarking Embedding- and Prompt-Based Model Performance on Semantic Similarity</title>
</titleInfo>
<name type="personal">
<namePart type="given">Lizzy</namePart>
<namePart type="family">Brans</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jelke</namePart>
<namePart type="family">Bloem</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We introduce Dutch Multi-SimLex, a 1,888–pair extension of the Multi-SimLex benchmark for evaluating lexical semantic similarity in Dutch. The dataset was rated by 100 native speakers on a 0–6 scale and shows high reliability (overall ICC(2,k)=0.82) as well as strong alignment with English (\ensuremathρ=0.73). Using this resource, we evaluate eighteen models across four architectural families: static embeddings, encoder-only transformers, encoder–decoders, and decoder-only LLMs. We evaluate models using two complementary approaches: embedding-based cosine similarity and prompted similarity judgments in Dutch. In embedding-based evaluation, FastText (\ensuremathρ=0.485) and the monolingual Dutch encoder BERTje (\ensuremathρ=0.468) achieve the strongest alignment with human ratings, while multilingual encoders such as mBERT (\ensuremathρ=0.208) and XLM-R (\ensuremathρ=0.186) perform weaker. Prompt-based evaluation yields substantially higher correlations, with GPT-4 (\ensuremathρ=0.761) performing best, followed by DeepSeek-V3 (\ensuremathρ=0.753) and Gemini 1.5 Pro (\ensuremathρ=0.722). Together, the results show that model performance depends strongly on how meaning is tested. Dutch Multi-SimLex provides a reliable foundation for evaluating meaning across architectures and advancing Dutch semantic evaluation.</abstract>
<identifier type="citekey">brans-bloem-2026-multi</identifier>
<identifier type="doi">10.63317/2q9dcx9cvnu9</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.380/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>4846</start>
<end>4860</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Multi-SimLex for Dutch: Benchmarking Embedding- and Prompt-Based Model Performance on Semantic Similarity
%A Brans, Lizzy
%A Bloem, Jelke
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F brans-bloem-2026-multi
%X We introduce Dutch Multi-SimLex, a 1,888–pair extension of the Multi-SimLex benchmark for evaluating lexical semantic similarity in Dutch. The dataset was rated by 100 native speakers on a 0–6 scale and shows high reliability (overall ICC(2,k)=0.82) as well as strong alignment with English (\ensuremathρ=0.73). Using this resource, we evaluate eighteen models across four architectural families: static embeddings, encoder-only transformers, encoder–decoders, and decoder-only LLMs. We evaluate models using two complementary approaches: embedding-based cosine similarity and prompted similarity judgments in Dutch. In embedding-based evaluation, FastText (\ensuremathρ=0.485) and the monolingual Dutch encoder BERTje (\ensuremathρ=0.468) achieve the strongest alignment with human ratings, while multilingual encoders such as mBERT (\ensuremathρ=0.208) and XLM-R (\ensuremathρ=0.186) perform weaker. Prompt-based evaluation yields substantially higher correlations, with GPT-4 (\ensuremathρ=0.761) performing best, followed by DeepSeek-V3 (\ensuremathρ=0.753) and Gemini 1.5 Pro (\ensuremathρ=0.722). Together, the results show that model performance depends strongly on how meaning is tested. Dutch Multi-SimLex provides a reliable foundation for evaluating meaning across architectures and advancing Dutch semantic evaluation.
%R 10.63317/2q9dcx9cvnu9
%U https://aclanthology.org/2026.lrec-1.380/
%U https://doi.org/10.63317/2q9dcx9cvnu9
%P 4846-4860
Markdown (Informal)
[Multi-SimLex for Dutch: Benchmarking Embedding- and Prompt-Based Model Performance on Semantic Similarity](https://aclanthology.org/2026.lrec-1.380/) (Brans & Bloem, LREC 2026)
ACL