@inproceedings{sakhawat-etal-2026-words,
title = "When Words Don{'}t Mean What They Say: Figurative Understanding in {B}engali Idioms",
author = "Sakhawat, Adib and
Parveen, Shamim Ara and
Amin, Md Ruhul and
Khatun, Tahera and
Mahmud, Shamim Al and
Islam, Md Saiful",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.546/",
doi = "10.63317/546w2cys6m6t",
pages = "6870--6879",
abstract = "Figurative language understanding remains a significant challenge for Large Language Models (LLMs), especially for low-resource languages. To address this, we introduce the \textit{Bangla Bagdhara} dataset, a large-scale, culturally grounded corpus of 10,361 Bengali idioms. Each idiom is annotated under a comprehensive 19-field schema, established and refined through a deliberative expert consensus process that captures its semantic, syntactic, cultural, and religious dimensions, providing a rich and structured resource for computational linguistics. To establish a robust benchmark for Bangla figurative language understanding, we evaluate 30 state-of-the-art multilingual and instruction-tuned LLMs on the task of inferring figurative meaning. Our results reveal a critical performance gap, with no model surpassing 50{\%} accuracy, in stark contrast to significantly higher human performance (83.4{\%}). This finding underscores the limitations of existing models in cross-linguistic and cultural reasoning. By releasing the \textit{Bangla Bagdhara} dataset and benchmark, we provide foundational infrastructure for advancing figurative language understanding and cultural grounding in LLMs for Bengali and other low-resource languages."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="sakhawat-etal-2026-words">
<titleInfo>
<title>When Words Don’t Mean What They Say: Figurative Understanding in Bengali Idioms</title>
</titleInfo>
<name type="personal">
<namePart type="given">Adib</namePart>
<namePart type="family">Sakhawat</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shamim</namePart>
<namePart type="given">Ara</namePart>
<namePart type="family">Parveen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Md</namePart>
<namePart type="given">Ruhul</namePart>
<namePart type="family">Amin</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tahera</namePart>
<namePart type="family">Khatun</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shamim</namePart>
<namePart type="given">Al</namePart>
<namePart type="family">Mahmud</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Md</namePart>
<namePart type="given">Saiful</namePart>
<namePart type="family">Islam</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Figurative language understanding remains a significant challenge for Large Language Models (LLMs), especially for low-resource languages. To address this, we introduce the Bangla Bagdhara dataset, a large-scale, culturally grounded corpus of 10,361 Bengali idioms. Each idiom is annotated under a comprehensive 19-field schema, established and refined through a deliberative expert consensus process that captures its semantic, syntactic, cultural, and religious dimensions, providing a rich and structured resource for computational linguistics. To establish a robust benchmark for Bangla figurative language understanding, we evaluate 30 state-of-the-art multilingual and instruction-tuned LLMs on the task of inferring figurative meaning. Our results reveal a critical performance gap, with no model surpassing 50% accuracy, in stark contrast to significantly higher human performance (83.4%). This finding underscores the limitations of existing models in cross-linguistic and cultural reasoning. By releasing the Bangla Bagdhara dataset and benchmark, we provide foundational infrastructure for advancing figurative language understanding and cultural grounding in LLMs for Bengali and other low-resource languages.</abstract>
<identifier type="citekey">sakhawat-etal-2026-words</identifier>
<identifier type="doi">10.63317/546w2cys6m6t</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.546/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>6870</start>
<end>6879</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T When Words Don’t Mean What They Say: Figurative Understanding in Bengali Idioms
%A Sakhawat, Adib
%A Parveen, Shamim Ara
%A Amin, Md Ruhul
%A Khatun, Tahera
%A Mahmud, Shamim Al
%A Islam, Md Saiful
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F sakhawat-etal-2026-words
%X Figurative language understanding remains a significant challenge for Large Language Models (LLMs), especially for low-resource languages. To address this, we introduce the Bangla Bagdhara dataset, a large-scale, culturally grounded corpus of 10,361 Bengali idioms. Each idiom is annotated under a comprehensive 19-field schema, established and refined through a deliberative expert consensus process that captures its semantic, syntactic, cultural, and religious dimensions, providing a rich and structured resource for computational linguistics. To establish a robust benchmark for Bangla figurative language understanding, we evaluate 30 state-of-the-art multilingual and instruction-tuned LLMs on the task of inferring figurative meaning. Our results reveal a critical performance gap, with no model surpassing 50% accuracy, in stark contrast to significantly higher human performance (83.4%). This finding underscores the limitations of existing models in cross-linguistic and cultural reasoning. By releasing the Bangla Bagdhara dataset and benchmark, we provide foundational infrastructure for advancing figurative language understanding and cultural grounding in LLMs for Bengali and other low-resource languages.
%R 10.63317/546w2cys6m6t
%U https://aclanthology.org/2026.lrec-1.546/
%U https://doi.org/10.63317/546w2cys6m6t
%P 6870-6879
Markdown (Informal)
[When Words Don’t Mean What They Say: Figurative Understanding in Bengali Idioms](https://aclanthology.org/2026.lrec-1.546/) (Sakhawat et al., LREC 2026)
ACL