@inproceedings{arslan-etal-2026-tr,
title = "{TR}-{TEB}: {T}urkish Text Embedding Benchmark",
author = "Arslan, Omer and
Celik, Atalay and
Aslan, Yusuf and
Durkaya, Hasan Fatih and
Zenginoglu, Mustafa Furkan and
Yilmaz, Musa Alperen and
Kantarci, Merve Gul and
Haklidir, Mehmet",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.862/",
doi = "10.63317/3qway8hn6y53",
pages = "11028--11044",
abstract = "Text embeddings are central to modern natural language processing, enabling several downstream tasks. Despite their significance, existing evaluation frameworks primarily target English and other high-resource languages, leaving critical gaps for languages such as Turkish. To address this, we present TR-TEB (Turkish Text Embedding Benchmark), the first comprehensive, standardized, and reproducible benchmark for Turkish text embeddings. TR-TEB spans five core task categories: classification, pair classification, clustering, retrieval, and semantic textual similarity. It is supported by a diverse dataset portfolio that integrates 14 curated open-source resources, 26 high-quality translated datasets, and 7 newly constructed Turkish-specific datasets designed to capture the language{'}s unique characteristics. We test our framework by comparing 45 well-known open-source embedding models. As the first unified evaluation suite, TR-TEB serves as a core tool for the Turkish embedding research community, establishing a systematic basis for model comparison and improvement. Furthermore, its benchmarking methodology and dataset creation process provide a blueprint for extending robust embedding evaluation to other low-resource languages."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="arslan-etal-2026-tr">
<titleInfo>
<title>TR-TEB: Turkish Text Embedding Benchmark</title>
</titleInfo>
<name type="personal">
<namePart type="given">Omer</namePart>
<namePart type="family">Arslan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Atalay</namePart>
<namePart type="family">Celik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yusuf</namePart>
<namePart type="family">Aslan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hasan</namePart>
<namePart type="given">Fatih</namePart>
<namePart type="family">Durkaya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mustafa</namePart>
<namePart type="given">Furkan</namePart>
<namePart type="family">Zenginoglu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Musa</namePart>
<namePart type="given">Alperen</namePart>
<namePart type="family">Yilmaz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Merve</namePart>
<namePart type="given">Gul</namePart>
<namePart type="family">Kantarci</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mehmet</namePart>
<namePart type="family">Haklidir</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Text embeddings are central to modern natural language processing, enabling several downstream tasks. Despite their significance, existing evaluation frameworks primarily target English and other high-resource languages, leaving critical gaps for languages such as Turkish. To address this, we present TR-TEB (Turkish Text Embedding Benchmark), the first comprehensive, standardized, and reproducible benchmark for Turkish text embeddings. TR-TEB spans five core task categories: classification, pair classification, clustering, retrieval, and semantic textual similarity. It is supported by a diverse dataset portfolio that integrates 14 curated open-source resources, 26 high-quality translated datasets, and 7 newly constructed Turkish-specific datasets designed to capture the language’s unique characteristics. We test our framework by comparing 45 well-known open-source embedding models. As the first unified evaluation suite, TR-TEB serves as a core tool for the Turkish embedding research community, establishing a systematic basis for model comparison and improvement. Furthermore, its benchmarking methodology and dataset creation process provide a blueprint for extending robust embedding evaluation to other low-resource languages.</abstract>
<identifier type="citekey">arslan-etal-2026-tr</identifier>
<identifier type="doi">10.63317/3qway8hn6y53</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.862/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>11028</start>
<end>11044</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T TR-TEB: Turkish Text Embedding Benchmark
%A Arslan, Omer
%A Celik, Atalay
%A Aslan, Yusuf
%A Durkaya, Hasan Fatih
%A Zenginoglu, Mustafa Furkan
%A Yilmaz, Musa Alperen
%A Kantarci, Merve Gul
%A Haklidir, Mehmet
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F arslan-etal-2026-tr
%X Text embeddings are central to modern natural language processing, enabling several downstream tasks. Despite their significance, existing evaluation frameworks primarily target English and other high-resource languages, leaving critical gaps for languages such as Turkish. To address this, we present TR-TEB (Turkish Text Embedding Benchmark), the first comprehensive, standardized, and reproducible benchmark for Turkish text embeddings. TR-TEB spans five core task categories: classification, pair classification, clustering, retrieval, and semantic textual similarity. It is supported by a diverse dataset portfolio that integrates 14 curated open-source resources, 26 high-quality translated datasets, and 7 newly constructed Turkish-specific datasets designed to capture the language’s unique characteristics. We test our framework by comparing 45 well-known open-source embedding models. As the first unified evaluation suite, TR-TEB serves as a core tool for the Turkish embedding research community, establishing a systematic basis for model comparison and improvement. Furthermore, its benchmarking methodology and dataset creation process provide a blueprint for extending robust embedding evaluation to other low-resource languages.
%R 10.63317/3qway8hn6y53
%U https://aclanthology.org/2026.lrec-1.862/
%U https://doi.org/10.63317/3qway8hn6y53
%P 11028-11044
Markdown (Informal)
[TR-TEB: Turkish Text Embedding Benchmark](https://aclanthology.org/2026.lrec-1.862/) (Arslan et al., LREC 2026)
ACL
- Omer Arslan, Atalay Celik, Yusuf Aslan, Hasan Fatih Durkaya, Mustafa Furkan Zenginoglu, Musa Alperen Yilmaz, Merve Gul Kantarci, and Mehmet Haklidir. 2026. TR-TEB: Turkish Text Embedding Benchmark. In Proceedings of the Fifteenth Language Resources and Evaluation Conference, pages 11028–11044, Palma de Mallorca, Spain. ELRA Language Resource Association.