@article{merlo-etal-2026-common,
title = "Common Objects Out of Context ({COOC}o): Investigating Multimodal Context and Semantic Scene Violations in Referential Communication",
author = "Merlo, Filippo and
Takmaz, Ece and
Chen, Wenkai and
Gatt, Albert",
journal = "Transactions of the Association for Computational Linguistics",
volume = "14",
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.tacl-1.52/",
doi = "10.1162/tacl.a.701",
pages = "1163--1185",
abstract = "To what degree and under what conditions do VLMs rely on scene context when generating references to objects? To address this question, we introduce the Common Objects Out-of-Context (COOCo) dataset and conduct experiments on several VLMs under different degrees of scene{--}object congruency and noise. We find that models leverage scene context adaptively, depending on scene-object semantic relatedness and noise level. Based on these consistent trends across models, we turn to the question of how VLM attention patterns change as a function of target-scene semantic fit, and to what degree these patterns are predictive of categorisation accuracy. We find that successful object categorisation is associated with increased mid-layer attention to the target. We also find a non-monotonic dependency on semantic fit, with attention dropping at moderate fit and increasing for both low and high fit. These results suggest that VLMs dynamically balance local and contextual information for reference generation. Dataset and code are available here: https://github.com/cs-nlp-uu/scenereg."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="merlo-etal-2026-common">
<titleInfo>
<title>Common Objects Out of Context (COOCo): Investigating Multimodal Context and Semantic Scene Violations in Referential Communication</title>
</titleInfo>
<name type="personal">
<namePart type="given">Filippo</namePart>
<namePart type="family">Merlo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ece</namePart>
<namePart type="family">Takmaz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Wenkai</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Albert</namePart>
<namePart type="family">Gatt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Transactions of the Association for Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>To what degree and under what conditions do VLMs rely on scene context when generating references to objects? To address this question, we introduce the Common Objects Out-of-Context (COOCo) dataset and conduct experiments on several VLMs under different degrees of scene–object congruency and noise. We find that models leverage scene context adaptively, depending on scene-object semantic relatedness and noise level. Based on these consistent trends across models, we turn to the question of how VLM attention patterns change as a function of target-scene semantic fit, and to what degree these patterns are predictive of categorisation accuracy. We find that successful object categorisation is associated with increased mid-layer attention to the target. We also find a non-monotonic dependency on semantic fit, with attention dropping at moderate fit and increasing for both low and high fit. These results suggest that VLMs dynamically balance local and contextual information for reference generation. Dataset and code are available here: https://github.com/cs-nlp-uu/scenereg.</abstract>
<identifier type="citekey">merlo-etal-2026-common</identifier>
<identifier type="doi">10.1162/tacl.a.701</identifier>
<location>
<url>https://aclanthology.org/2026.tacl-1.52/</url>
</location>
<part>
<date>2026</date>
<detail type="volume"><number>14</number></detail>
<extent unit="page">
<start>1163</start>
<end>1185</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T Common Objects Out of Context (COOCo): Investigating Multimodal Context and Semantic Scene Violations in Referential Communication
%A Merlo, Filippo
%A Takmaz, Ece
%A Chen, Wenkai
%A Gatt, Albert
%J Transactions of the Association for Computational Linguistics
%D 2026
%V 14
%I MIT Press
%C Cambridge, MA
%F merlo-etal-2026-common
%X To what degree and under what conditions do VLMs rely on scene context when generating references to objects? To address this question, we introduce the Common Objects Out-of-Context (COOCo) dataset and conduct experiments on several VLMs under different degrees of scene–object congruency and noise. We find that models leverage scene context adaptively, depending on scene-object semantic relatedness and noise level. Based on these consistent trends across models, we turn to the question of how VLM attention patterns change as a function of target-scene semantic fit, and to what degree these patterns are predictive of categorisation accuracy. We find that successful object categorisation is associated with increased mid-layer attention to the target. We also find a non-monotonic dependency on semantic fit, with attention dropping at moderate fit and increasing for both low and high fit. These results suggest that VLMs dynamically balance local and contextual information for reference generation. Dataset and code are available here: https://github.com/cs-nlp-uu/scenereg.
%R 10.1162/tacl.a.701
%U https://aclanthology.org/2026.tacl-1.52/
%U https://doi.org/10.1162/tacl.a.701
%P 1163-1185
Markdown (Informal)
[Common Objects Out of Context (COOCo): Investigating Multimodal Context and Semantic Scene Violations in Referential Communication](https://aclanthology.org/2026.tacl-1.52/) (Merlo et al., TACL 2026)
ACL