@inproceedings{coltekin-gunes-2026-corpus,
title = "A Corpus of Misunderstood Irony on {T}urkish Social Media",
author = {{\c{C}}{\"o}ltekin, {\c{C}}a{\u{g}}r{\i} and
G{\"u}ne{\c{s}}, G{\"u}liz},
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.879/",
doi = "10.63317/3kehaa7yjjqc",
pages = "11252--11259",
abstract = "We present a new Turkish social media corpus annotated for verbal irony. The ironic post candidates are identified by a distant supervision method relying on reports of misunderstood irony in social media platforms. The data collected through this method, as well as irony-tagged posts and a random sample of posts are annotated by three annotators, resulting in a corpus of 3000 tweets with high quality annotations that may be useful for linguistic analysis as well as for training automatic irony detection systems or testing irony understanding of large language models. Since irony interpretation typically involves context, our dataset also includes the preceding conversational context of the potentially ironic expression. Besides the description of the corpus and the annotation process, this paper presents an analysis of the corpus. Our findings indicate that relying on distant supervision alone may result in suboptimal labels for irony/sarcasm corpora. We also investigate the usefulness of context for the annotators in identifying irony."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="coltekin-gunes-2026-corpus">
<titleInfo>
<title>A Corpus of Misunderstood Irony on Turkish Social Media</title>
</titleInfo>
<name type="personal">
<namePart type="given">Çağrı</namePart>
<namePart type="family">Çöltekin</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Güliz</namePart>
<namePart type="family">Güneş</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present a new Turkish social media corpus annotated for verbal irony. The ironic post candidates are identified by a distant supervision method relying on reports of misunderstood irony in social media platforms. The data collected through this method, as well as irony-tagged posts and a random sample of posts are annotated by three annotators, resulting in a corpus of 3000 tweets with high quality annotations that may be useful for linguistic analysis as well as for training automatic irony detection systems or testing irony understanding of large language models. Since irony interpretation typically involves context, our dataset also includes the preceding conversational context of the potentially ironic expression. Besides the description of the corpus and the annotation process, this paper presents an analysis of the corpus. Our findings indicate that relying on distant supervision alone may result in suboptimal labels for irony/sarcasm corpora. We also investigate the usefulness of context for the annotators in identifying irony.</abstract>
<identifier type="citekey">coltekin-gunes-2026-corpus</identifier>
<identifier type="doi">10.63317/3kehaa7yjjqc</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.879/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>11252</start>
<end>11259</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T A Corpus of Misunderstood Irony on Turkish Social Media
%A Çöltekin, Çağrı
%A Güneş, Güliz
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F coltekin-gunes-2026-corpus
%X We present a new Turkish social media corpus annotated for verbal irony. The ironic post candidates are identified by a distant supervision method relying on reports of misunderstood irony in social media platforms. The data collected through this method, as well as irony-tagged posts and a random sample of posts are annotated by three annotators, resulting in a corpus of 3000 tweets with high quality annotations that may be useful for linguistic analysis as well as for training automatic irony detection systems or testing irony understanding of large language models. Since irony interpretation typically involves context, our dataset also includes the preceding conversational context of the potentially ironic expression. Besides the description of the corpus and the annotation process, this paper presents an analysis of the corpus. Our findings indicate that relying on distant supervision alone may result in suboptimal labels for irony/sarcasm corpora. We also investigate the usefulness of context for the annotators in identifying irony.
%R 10.63317/3kehaa7yjjqc
%U https://aclanthology.org/2026.lrec-1.879/
%U https://doi.org/10.63317/3kehaa7yjjqc
%P 11252-11259
Markdown (Informal)
[A Corpus of Misunderstood Irony on Turkish Social Media](https://aclanthology.org/2026.lrec-1.879/) (Çöltekin & Güneş, LREC 2026)
ACL