@inproceedings{larsson-etal-2026-visual,
title = "Visual Question Answering and the relation between language, perception and the world",
author = "Larsson, Staffan and
Noble, Bill and
Cooper, Robin",
editor = "Yanaka, Hitomi and
Abzianidze, Lasha",
booktitle = "Proceedings of the 6th Workshop on Natural Language Meets Logic and Machine Learning ({NALOMA})",
month = aug,
year = "2026",
address = "Prague, Czechia",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.naloma-1.8/",
pages = "60--70",
ISBN = "979-8-89176-389-0",
abstract = "We outline a general account of VQA using a perception-oriented formal semantic framework. We believe that it is instructive to describe VQA not only in terms of an engineering challenge, but also as a linguistically and philosophically relevant task that can help us better understand the relation between language, perception and the world. Specifically, we will argue that linguistic meaning helps us structure our takes on visual scenes, enabling us to classify situations so that we e.g. can answer questions and determine whether a sentence correctly describes a scene."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="larsson-etal-2026-visual">
<titleInfo>
<title>Visual Question Answering and the relation between language, perception and the world</title>
</titleInfo>
<name type="personal">
<namePart type="given">Staffan</namePart>
<namePart type="family">Larsson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Bill</namePart>
<namePart type="family">Noble</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Robin</namePart>
<namePart type="family">Cooper</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-08</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 6th Workshop on Natural Language Meets Logic and Machine Learning (NALOMA)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hitomi</namePart>
<namePart type="family">Yanaka</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lasha</namePart>
<namePart type="family">Abzianidze</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Prague, Czechia</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-89176-389-0</identifier>
</relatedItem>
<abstract>We outline a general account of VQA using a perception-oriented formal semantic framework. We believe that it is instructive to describe VQA not only in terms of an engineering challenge, but also as a linguistically and philosophically relevant task that can help us better understand the relation between language, perception and the world. Specifically, we will argue that linguistic meaning helps us structure our takes on visual scenes, enabling us to classify situations so that we e.g. can answer questions and determine whether a sentence correctly describes a scene.</abstract>
<identifier type="citekey">larsson-etal-2026-visual</identifier>
<location>
<url>https://aclanthology.org/2026.naloma-1.8/</url>
</location>
<part>
<date>2026-08</date>
<extent unit="page">
<start>60</start>
<end>70</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Visual Question Answering and the relation between language, perception and the world
%A Larsson, Staffan
%A Noble, Bill
%A Cooper, Robin
%Y Yanaka, Hitomi
%Y Abzianidze, Lasha
%S Proceedings of the 6th Workshop on Natural Language Meets Logic and Machine Learning (NALOMA)
%D 2026
%8 August
%I Association for Computational Linguistics
%C Prague, Czechia
%@ 979-8-89176-389-0
%F larsson-etal-2026-visual
%X We outline a general account of VQA using a perception-oriented formal semantic framework. We believe that it is instructive to describe VQA not only in terms of an engineering challenge, but also as a linguistically and philosophically relevant task that can help us better understand the relation between language, perception and the world. Specifically, we will argue that linguistic meaning helps us structure our takes on visual scenes, enabling us to classify situations so that we e.g. can answer questions and determine whether a sentence correctly describes a scene.
%U https://aclanthology.org/2026.naloma-1.8/
%P 60-70
Markdown (Informal)
[Visual Question Answering and the relation between language, perception and the world](https://aclanthology.org/2026.naloma-1.8/) (Larsson et al., NALOMA 2026)
ACL