@inproceedings{cheng-utsuro-2026-llms,
title = "Can {LLM}s Understand Punchlines? {LLM}s' Narrative Understanding Evaluation with Short-shorts",
author = "Cheng, Jiashi and
Utsuro, Takehito",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.159/",
doi = "10.63317/4n2p36736i24",
pages = "2024--2034",
abstract = "In this study, we constructed a narrative comprehension benchmark using the works of Shinichi Hoshi to examine the extent to which Large Language Models (LLMs) can understand twist endings, or punchlines, in short-short stories. Specifically, story endings were categorized into six types{---}such as Revelation, Apocalypse, and Sarcasm{---}and a classification task was designed in which LLMs were prompted with the story text and asked to select the appropriate ending category. We collected human annotations from eight native Japanese speakers to establish a reference benchmark. Experimental comparisons were conducted across multiple LLMs (GPT-4, Claude, Gemini, and Grok), assessing their performance both at the metric level and at the discourse level against human judgments. The results revealed that although certain models approached human performance in specific categories, overall accuracy remained notably lower than the human baseline. Through quantitative and qualitative analyses, this study highlights the challenges LLMs face in capturing narrative subtleties such as irony, implication, and emotional reversal. The proposed benchmark provides a novel framework for evaluating narrative understanding and the deeper semantic reasoning capabilities of LLMs."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="cheng-utsuro-2026-llms">
<titleInfo>
<title>Can LLMs Understand Punchlines? LLMs’ Narrative Understanding Evaluation with Short-shorts</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jiashi</namePart>
<namePart type="family">Cheng</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Takehito</namePart>
<namePart type="family">Utsuro</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>In this study, we constructed a narrative comprehension benchmark using the works of Shinichi Hoshi to examine the extent to which Large Language Models (LLMs) can understand twist endings, or punchlines, in short-short stories. Specifically, story endings were categorized into six types—such as Revelation, Apocalypse, and Sarcasm—and a classification task was designed in which LLMs were prompted with the story text and asked to select the appropriate ending category. We collected human annotations from eight native Japanese speakers to establish a reference benchmark. Experimental comparisons were conducted across multiple LLMs (GPT-4, Claude, Gemini, and Grok), assessing their performance both at the metric level and at the discourse level against human judgments. The results revealed that although certain models approached human performance in specific categories, overall accuracy remained notably lower than the human baseline. Through quantitative and qualitative analyses, this study highlights the challenges LLMs face in capturing narrative subtleties such as irony, implication, and emotional reversal. The proposed benchmark provides a novel framework for evaluating narrative understanding and the deeper semantic reasoning capabilities of LLMs.</abstract>
<identifier type="citekey">cheng-utsuro-2026-llms</identifier>
<identifier type="doi">10.63317/4n2p36736i24</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.159/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>2024</start>
<end>2034</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Can LLMs Understand Punchlines? LLMs’ Narrative Understanding Evaluation with Short-shorts
%A Cheng, Jiashi
%A Utsuro, Takehito
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F cheng-utsuro-2026-llms
%X In this study, we constructed a narrative comprehension benchmark using the works of Shinichi Hoshi to examine the extent to which Large Language Models (LLMs) can understand twist endings, or punchlines, in short-short stories. Specifically, story endings were categorized into six types—such as Revelation, Apocalypse, and Sarcasm—and a classification task was designed in which LLMs were prompted with the story text and asked to select the appropriate ending category. We collected human annotations from eight native Japanese speakers to establish a reference benchmark. Experimental comparisons were conducted across multiple LLMs (GPT-4, Claude, Gemini, and Grok), assessing their performance both at the metric level and at the discourse level against human judgments. The results revealed that although certain models approached human performance in specific categories, overall accuracy remained notably lower than the human baseline. Through quantitative and qualitative analyses, this study highlights the challenges LLMs face in capturing narrative subtleties such as irony, implication, and emotional reversal. The proposed benchmark provides a novel framework for evaluating narrative understanding and the deeper semantic reasoning capabilities of LLMs.
%R 10.63317/4n2p36736i24
%U https://aclanthology.org/2026.lrec-1.159/
%U https://doi.org/10.63317/4n2p36736i24
%P 2024-2034
Markdown (Informal)
[Can LLMs Understand Punchlines? LLMs’ Narrative Understanding Evaluation with Short-shorts](https://aclanthology.org/2026.lrec-1.159/) (Cheng & Utsuro, LREC 2026)
ACL