@inproceedings{niloy-etal-2026-meaning,
title = "Meaning Over Morphology: A Multi-Metric Benchmark of {LLM}s for {B}angla Dialect Translation",
author = "Niloy, Soumik Deb and
Rahman, Subhey Sadi and
E Sobhani, Mahbub and
Alam, Md. Golam Rabiul and
Sadeque, Farig Yousuf and
Hassan, Md. Rezuwan",
editor = "Anastasopoulos, Antonis and
Markantonatou, Stella and
Ralli, Angela and
Zampieri, Marcos and
Bompolas, Stavros and
Stamou, Vivian",
booktitle = "Proceedings of the First Workshop on Dialects in {NLP} {---} A Resource Perspective",
month = may,
year = "2026",
address = "Palma de Mallorca",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.dialres-1.24/",
doi = "10.63317/3h8y7fs2zbq4",
pages = "238--255",
abstract = "Regional dialects of Bangla, such as Sylheti and Chittagonian, pose significant challenges for natural language processing due to their low-resource nature and substantial linguistic variation from standard Bangla. In this work, we present a systematic evaluation of eight open-source LLMs for translating fifteen distinct Bangla dialects into standard Bangla. To achieve this comprehensive coverage, we utilize a combination of established benchmarks and a novel dataset curated from an ongoing regional linguistic project. We assess model performance using a multi-metric framework that combines exact-match and error-rate evaluations such as, Averaged BLEU, WER, and CER with embedding-based semantic metrics including BERTScore, METEOR, and COMET. Additionally, we perform a detailed dialect-level linguistic analysis to identify the deep-seated structural, orthographic, and semantic barriers inherent to dialectal translation. Our study highlights the strengths and limitations of current open-source models, provides empirical insights for future dialect-aware fine-tuning, and contributes a reproducible benchmark for the research community."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="niloy-etal-2026-meaning">
<titleInfo>
<title>Meaning Over Morphology: A Multi-Metric Benchmark of LLMs for Bangla Dialect Translation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Soumik</namePart>
<namePart type="given">Deb</namePart>
<namePart type="family">Niloy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Subhey</namePart>
<namePart type="given">Sadi</namePart>
<namePart type="family">Rahman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mahbub</namePart>
<namePart type="family">E Sobhani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Md.</namePart>
<namePart type="given">Golam</namePart>
<namePart type="given">Rabiul</namePart>
<namePart type="family">Alam</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Farig</namePart>
<namePart type="given">Yousuf</namePart>
<namePart type="family">Sadeque</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Md.</namePart>
<namePart type="given">Rezuwan</namePart>
<namePart type="family">Hassan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective</title>
</titleInfo>
<name type="personal">
<namePart type="given">Antonis</namePart>
<namePart type="family">Anastasopoulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stella</namePart>
<namePart type="family">Markantonatou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Angela</namePart>
<namePart type="family">Ralli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marcos</namePart>
<namePart type="family">Zampieri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stavros</namePart>
<namePart type="family">Bompolas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vivian</namePart>
<namePart type="family">Stamou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Regional dialects of Bangla, such as Sylheti and Chittagonian, pose significant challenges for natural language processing due to their low-resource nature and substantial linguistic variation from standard Bangla. In this work, we present a systematic evaluation of eight open-source LLMs for translating fifteen distinct Bangla dialects into standard Bangla. To achieve this comprehensive coverage, we utilize a combination of established benchmarks and a novel dataset curated from an ongoing regional linguistic project. We assess model performance using a multi-metric framework that combines exact-match and error-rate evaluations such as, Averaged BLEU, WER, and CER with embedding-based semantic metrics including BERTScore, METEOR, and COMET. Additionally, we perform a detailed dialect-level linguistic analysis to identify the deep-seated structural, orthographic, and semantic barriers inherent to dialectal translation. Our study highlights the strengths and limitations of current open-source models, provides empirical insights for future dialect-aware fine-tuning, and contributes a reproducible benchmark for the research community.</abstract>
<identifier type="citekey">niloy-etal-2026-meaning</identifier>
<identifier type="doi">10.63317/3h8y7fs2zbq4</identifier>
<location>
<url>https://aclanthology.org/2026.dialres-1.24/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>238</start>
<end>255</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Meaning Over Morphology: A Multi-Metric Benchmark of LLMs for Bangla Dialect Translation
%A Niloy, Soumik Deb
%A Rahman, Subhey Sadi
%A E Sobhani, Mahbub
%A Alam, Md. Golam Rabiul
%A Sadeque, Farig Yousuf
%A Hassan, Md. Rezuwan
%Y Anastasopoulos, Antonis
%Y Markantonatou, Stella
%Y Ralli, Angela
%Y Zampieri, Marcos
%Y Bompolas, Stavros
%Y Stamou, Vivian
%S Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca
%F niloy-etal-2026-meaning
%X Regional dialects of Bangla, such as Sylheti and Chittagonian, pose significant challenges for natural language processing due to their low-resource nature and substantial linguistic variation from standard Bangla. In this work, we present a systematic evaluation of eight open-source LLMs for translating fifteen distinct Bangla dialects into standard Bangla. To achieve this comprehensive coverage, we utilize a combination of established benchmarks and a novel dataset curated from an ongoing regional linguistic project. We assess model performance using a multi-metric framework that combines exact-match and error-rate evaluations such as, Averaged BLEU, WER, and CER with embedding-based semantic metrics including BERTScore, METEOR, and COMET. Additionally, we perform a detailed dialect-level linguistic analysis to identify the deep-seated structural, orthographic, and semantic barriers inherent to dialectal translation. Our study highlights the strengths and limitations of current open-source models, provides empirical insights for future dialect-aware fine-tuning, and contributes a reproducible benchmark for the research community.
%R 10.63317/3h8y7fs2zbq4
%U https://aclanthology.org/2026.dialres-1.24/
%U https://doi.org/10.63317/3h8y7fs2zbq4
%P 238-255
Markdown (Informal)
[Meaning Over Morphology: A Multi-Metric Benchmark of LLMs for Bangla Dialect Translation](https://aclanthology.org/2026.dialres-1.24/) (Niloy et al., DialRes 2026)
ACL