@inproceedings{gan-etal-2026-comicvqa,
title = "{C}omic{VQA}: A Benchmark for Visual Reasoning in Multimodal {LLM}s",
author = "Gan, Esther and
Brown, Hannah and
Herel, David and
Kawaguchi, Kenji and
Kan, Min-Yen and
Shieh, Michael Qizhe",
editor = "Liakata, Maria and
Moreira, Viviane P. and
Zhang, Jiajun and
Jurgens, David",
booktitle = "Findings of the {A}ssociation for {C}omputational {L}inguistics: {ACL} 2026",
month = jul,
year = "2026",
address = "San Diego, California, United States",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.findings-acl.1268/",
doi = "10.18653/v1/2026.findings-acl.1268",
pages = "25347--25370",
ISBN = "979-8-89176-395-1",
abstract = "We introduce Comic Visual Question Answering (\textbf{ComicVQA}), a comics-based benchmark for evaluating MLLMs on visual reasoning. ComicVQA comprises of (i) \textbf{Missing Panel Prediction}, testing fine-grained visual grounding and (ii) \textbf{Panel Sorting}, which evaluates sequential narrative understanding. Proprietary models achieve up to 62.6{\%} on Missing Panel Prediction and 46.4{\%} on Panel Sorting, whereas open-source models reach only 47.7{\%} and 26.9{\%}, respectively. In contrast, human annotators achieve over 83{\%} accuracy on both tasks, revealing a large gap between current models and human-level multimodal understanding in comics. Through controlled ordering ablations and a detailed error taxonomy, we show that current MLLMs rely primarily on coarse temporal cues and struggle with fine-grained visual reasoning. These findings demonstrate ComicVQA as a diagnostic benchmark for advancing multimodal visual reasoning in comics."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="gan-etal-2026-comicvqa">
<titleInfo>
<title>ComicVQA: A Benchmark for Visual Reasoning in Multimodal LLMs</title>
</titleInfo>
<name type="personal">
<namePart type="given">Esther</namePart>
<namePart type="family">Gan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hannah</namePart>
<namePart type="family">Brown</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">David</namePart>
<namePart type="family">Herel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kenji</namePart>
<namePart type="family">Kawaguchi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Min-Yen</namePart>
<namePart type="family">Kan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Michael</namePart>
<namePart type="given">Qizhe</namePart>
<namePart type="family">Shieh</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-07</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Findings of the Association for Computational Linguistics: ACL 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="family">Liakata</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Viviane</namePart>
<namePart type="given">P</namePart>
<namePart type="family">Moreira</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jiajun</namePart>
<namePart type="family">Zhang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">David</namePart>
<namePart type="family">Jurgens</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">San Diego, California, United States</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-89176-395-1</identifier>
</relatedItem>
<abstract>We introduce Comic Visual Question Answering (ComicVQA), a comics-based benchmark for evaluating MLLMs on visual reasoning. ComicVQA comprises of (i) Missing Panel Prediction, testing fine-grained visual grounding and (ii) Panel Sorting, which evaluates sequential narrative understanding. Proprietary models achieve up to 62.6% on Missing Panel Prediction and 46.4% on Panel Sorting, whereas open-source models reach only 47.7% and 26.9%, respectively. In contrast, human annotators achieve over 83% accuracy on both tasks, revealing a large gap between current models and human-level multimodal understanding in comics. Through controlled ordering ablations and a detailed error taxonomy, we show that current MLLMs rely primarily on coarse temporal cues and struggle with fine-grained visual reasoning. These findings demonstrate ComicVQA as a diagnostic benchmark for advancing multimodal visual reasoning in comics.</abstract>
<identifier type="citekey">gan-etal-2026-comicvqa</identifier>
<identifier type="doi">10.18653/v1/2026.findings-acl.1268</identifier>
<location>
<url>https://aclanthology.org/2026.findings-acl.1268/</url>
</location>
<part>
<date>2026-07</date>
<extent unit="page">
<start>25347</start>
<end>25370</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T ComicVQA: A Benchmark for Visual Reasoning in Multimodal LLMs
%A Gan, Esther
%A Brown, Hannah
%A Herel, David
%A Kawaguchi, Kenji
%A Kan, Min-Yen
%A Shieh, Michael Qizhe
%Y Liakata, Maria
%Y Moreira, Viviane P.
%Y Zhang, Jiajun
%Y Jurgens, David
%S Findings of the Association for Computational Linguistics: ACL 2026
%D 2026
%8 July
%I Association for Computational Linguistics
%C San Diego, California, United States
%@ 979-8-89176-395-1
%F gan-etal-2026-comicvqa
%X We introduce Comic Visual Question Answering (ComicVQA), a comics-based benchmark for evaluating MLLMs on visual reasoning. ComicVQA comprises of (i) Missing Panel Prediction, testing fine-grained visual grounding and (ii) Panel Sorting, which evaluates sequential narrative understanding. Proprietary models achieve up to 62.6% on Missing Panel Prediction and 46.4% on Panel Sorting, whereas open-source models reach only 47.7% and 26.9%, respectively. In contrast, human annotators achieve over 83% accuracy on both tasks, revealing a large gap between current models and human-level multimodal understanding in comics. Through controlled ordering ablations and a detailed error taxonomy, we show that current MLLMs rely primarily on coarse temporal cues and struggle with fine-grained visual reasoning. These findings demonstrate ComicVQA as a diagnostic benchmark for advancing multimodal visual reasoning in comics.
%R 10.18653/v1/2026.findings-acl.1268
%U https://aclanthology.org/2026.findings-acl.1268/
%U https://doi.org/10.18653/v1/2026.findings-acl.1268
%P 25347-25370
Markdown (Informal)
[ComicVQA: A Benchmark for Visual Reasoning in Multimodal LLMs](https://aclanthology.org/2026.findings-acl.1268/) (Gan et al., Findings 2026)
ACL
- Esther Gan, Hannah Brown, David Herel, Kenji Kawaguchi, Min-Yen Kan, and Michael Qizhe Shieh. 2026. ComicVQA: A Benchmark for Visual Reasoning in Multimodal LLMs. In Findings of the Association for Computational Linguistics: ACL 2026, pages 25347–25370, San Diego, California, United States. Association for Computational Linguistics.