@inproceedings{samajdar-2026-indeuph,
title = "{I}nd{E}uph-170: Benchmarking Cultural Pragmatics through Euphemism Detection in {I}ndian {E}nglish",
author = "Samajdar, Debamita",
editor = "Jha, Girish Nath and
Bali, Kalika and
L, Sobha and
Kumar, Devendr",
booktitle = "Proceedings of the 8th Workshop on {I}ndian Language Data: Resources and Evaluation",
month = may,
year = "2026",
address = "Palma, Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.wildre-1.12/",
doi = "10.63317/3ecgg3aq3dbk",
pages = "93--97",
abstract = "Large Language Models (LLMs) have shown remarkable proficiency in standard English benchmarks, yet their ability to navigate the sociopragmatic cues of non-Western English varieties remains underexplored. This paper introduces IndEuph-170, a novel benchmark dataset focused on Indian English (IndE) euphemisms {---} expressions whose roots lie in local social hierarchies, politeness norms, and cultural taboos (e.g., ``setting,'' ``loose character,'' ``suitable boy''). IndEuph-170 comprises 170 curated IndE sentences, against which the performance of two distinct architectures was evaluated: a fine-tuned BART model and GPT-4. The findings reveal a significant ``cultural gap''. While GPT-4 achieves 82.5{\%} accuracy, it struggles with authoritative and punitive nuances. BART achieves 55.3{\%} accuracy but exhibits a high rate of false positives by over-classifying general Indianisms as euphemisms. The paper argues that current multilingual benchmarks such as MME (Fu et al., 2025) and GLUE (Wang et al., 2018) fail to capture these dialectal pragmatics, and that a culturally-aware evaluation framework for Global Englishes is necessary."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="samajdar-2026-indeuph">
<titleInfo>
<title>IndEuph-170: Benchmarking Cultural Pragmatics through Euphemism Detection in Indian English</title>
</titleInfo>
<name type="personal">
<namePart type="given">Debamita</namePart>
<namePart type="family">Samajdar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 8th Workshop on Indian Language Data: Resources and Evaluation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Girish</namePart>
<namePart type="given">Nath</namePart>
<namePart type="family">Jha</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kalika</namePart>
<namePart type="family">Bali</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sobha</namePart>
<namePart type="family">L</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Devendr</namePart>
<namePart type="family">Kumar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Large Language Models (LLMs) have shown remarkable proficiency in standard English benchmarks, yet their ability to navigate the sociopragmatic cues of non-Western English varieties remains underexplored. This paper introduces IndEuph-170, a novel benchmark dataset focused on Indian English (IndE) euphemisms — expressions whose roots lie in local social hierarchies, politeness norms, and cultural taboos (e.g., “setting,” “loose character,” “suitable boy”). IndEuph-170 comprises 170 curated IndE sentences, against which the performance of two distinct architectures was evaluated: a fine-tuned BART model and GPT-4. The findings reveal a significant “cultural gap”. While GPT-4 achieves 82.5% accuracy, it struggles with authoritative and punitive nuances. BART achieves 55.3% accuracy but exhibits a high rate of false positives by over-classifying general Indianisms as euphemisms. The paper argues that current multilingual benchmarks such as MME (Fu et al., 2025) and GLUE (Wang et al., 2018) fail to capture these dialectal pragmatics, and that a culturally-aware evaluation framework for Global Englishes is necessary.</abstract>
<identifier type="citekey">samajdar-2026-indeuph</identifier>
<identifier type="doi">10.63317/3ecgg3aq3dbk</identifier>
<location>
<url>https://aclanthology.org/2026.wildre-1.12/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>93</start>
<end>97</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T IndEuph-170: Benchmarking Cultural Pragmatics through Euphemism Detection in Indian English
%A Samajdar, Debamita
%Y Jha, Girish Nath
%Y Bali, Kalika
%Y L, Sobha
%Y Kumar, Devendr
%S Proceedings of the 8th Workshop on Indian Language Data: Resources and Evaluation
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca, Spain
%F samajdar-2026-indeuph
%X Large Language Models (LLMs) have shown remarkable proficiency in standard English benchmarks, yet their ability to navigate the sociopragmatic cues of non-Western English varieties remains underexplored. This paper introduces IndEuph-170, a novel benchmark dataset focused on Indian English (IndE) euphemisms — expressions whose roots lie in local social hierarchies, politeness norms, and cultural taboos (e.g., “setting,” “loose character,” “suitable boy”). IndEuph-170 comprises 170 curated IndE sentences, against which the performance of two distinct architectures was evaluated: a fine-tuned BART model and GPT-4. The findings reveal a significant “cultural gap”. While GPT-4 achieves 82.5% accuracy, it struggles with authoritative and punitive nuances. BART achieves 55.3% accuracy but exhibits a high rate of false positives by over-classifying general Indianisms as euphemisms. The paper argues that current multilingual benchmarks such as MME (Fu et al., 2025) and GLUE (Wang et al., 2018) fail to capture these dialectal pragmatics, and that a culturally-aware evaluation framework for Global Englishes is necessary.
%R 10.63317/3ecgg3aq3dbk
%U https://aclanthology.org/2026.wildre-1.12/
%U https://doi.org/10.63317/3ecgg3aq3dbk
%P 93-97
Markdown (Informal)
[IndEuph-170: Benchmarking Cultural Pragmatics through Euphemism Detection in Indian English](https://aclanthology.org/2026.wildre-1.12/) (Samajdar, WILDRE 2026)
ACL