@inproceedings{guerra-khrapunova-2026-picture,
title = "Is a Picture Worth a Thousand Words? Exploration and Implementation Considerations for Visual Context in Translation Workflows",
author = "Guerra, Vera Senderowicz and
Khrapunova, Olesia",
editor = "Shterionov, Dimitar and
Vanmassenhove, Eva and
De Sisto, Mirella and
Blain, Fred and
Pourmostafa Roshan Sharami, Javad and
Lepp, Lisa and
Manna, Chiara and
Rescigno, Argentina Anna and
Karakanta, Alina and
Rigouts Terryn, Ayla and
Lardelli, Manuel and
Resende, Natalia and
Murgolo, Elena and
Hackenbuchner, Jani{\c{c}}a and
Zaretskaya, Anna and
Espl{\`a}-Gomis, Miquel and
Etchegoyhen, Thierry and
Gromann, Dagmar and
Bawden, Rachel and
Haddow, Barry and
Szoc, Sara and
Forcada, Mikel and
Moniz, Helena",
booktitle = "Proceedings of the 26th Annual Conference of the {E}uropean Association for Machine Translation (Volume 2)",
month = jun,
year = "2026",
address = "Tilburg, The Netherlands",
publisher = "European Association for Machine Translation",
url = "https://aclanthology.org/2026.eamt-2.29/",
pages = "91--103",
ISBN = "9789403901404",
abstract = "Vision-language models (VLMs) have the potential to enhance machine translation (MT) by leveraging visual context alongside text, yet their real utility for production workflows remains unclear. We conduct a unified, multi-condition evaluation of six leading VLMs{---}both open and proprietary{---}on two challenging benchmarks (CoMMuTE and CaMMT), targeting lexical and cultural disambiguation respectively, with a domain-style case study simulating technical documentation localization. Results show that model performance varies widely, and the benefit of relevant images does not necessarily transfer across use cases. Proprietary models are notably sensitive to irrelevant images while open-source models are generally more stable; incorrect or contradicting visuals, by contrast, degrade translation across all models. Taken together, these findings make rigorous evaluation a necessary precondition for production deployment: metric gains can mask real accuracy losses in technical domains, model sensitivity to irrelevant images should inform model selection, and reliable image{--}text matching is a hard requirement for any pipeline."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="guerra-khrapunova-2026-picture">
<titleInfo>
<title>Is a Picture Worth a Thousand Words? Exploration and Implementation Considerations for Visual Context in Translation Workflows</title>
</titleInfo>
<name type="personal">
<namePart type="given">Vera</namePart>
<namePart type="given">Senderowicz</namePart>
<namePart type="family">Guerra</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Olesia</namePart>
<namePart type="family">Khrapunova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 2)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Dimitar</namePart>
<namePart type="family">Shterionov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eva</namePart>
<namePart type="family">Vanmassenhove</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mirella</namePart>
<namePart type="family">De Sisto</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fred</namePart>
<namePart type="family">Blain</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Javad</namePart>
<namePart type="family">Pourmostafa Roshan Sharami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lisa</namePart>
<namePart type="family">Lepp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chiara</namePart>
<namePart type="family">Manna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Argentina</namePart>
<namePart type="given">Anna</namePart>
<namePart type="family">Rescigno</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Karakanta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ayla</namePart>
<namePart type="family">Rigouts Terryn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Manuel</namePart>
<namePart type="family">Lardelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Natalia</namePart>
<namePart type="family">Resende</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Murgolo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Janiça</namePart>
<namePart type="family">Hackenbuchner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anna</namePart>
<namePart type="family">Zaretskaya</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Miquel</namePart>
<namePart type="family">Esplà-Gomis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Thierry</namePart>
<namePart type="family">Etchegoyhen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dagmar</namePart>
<namePart type="family">Gromann</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rachel</namePart>
<namePart type="family">Bawden</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Barry</namePart>
<namePart type="family">Haddow</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sara</namePart>
<namePart type="family">Szoc</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mikel</namePart>
<namePart type="family">Forcada</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Helena</namePart>
<namePart type="family">Moniz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>European Association for Machine Translation</publisher>
<place>
<placeTerm type="text">Tilburg, The Netherlands</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">9789403901404</identifier>
</relatedItem>
<abstract>Vision-language models (VLMs) have the potential to enhance machine translation (MT) by leveraging visual context alongside text, yet their real utility for production workflows remains unclear. We conduct a unified, multi-condition evaluation of six leading VLMs—both open and proprietary—on two challenging benchmarks (CoMMuTE and CaMMT), targeting lexical and cultural disambiguation respectively, with a domain-style case study simulating technical documentation localization. Results show that model performance varies widely, and the benefit of relevant images does not necessarily transfer across use cases. Proprietary models are notably sensitive to irrelevant images while open-source models are generally more stable; incorrect or contradicting visuals, by contrast, degrade translation across all models. Taken together, these findings make rigorous evaluation a necessary precondition for production deployment: metric gains can mask real accuracy losses in technical domains, model sensitivity to irrelevant images should inform model selection, and reliable image–text matching is a hard requirement for any pipeline.</abstract>
<identifier type="citekey">guerra-khrapunova-2026-picture</identifier>
<location>
<url>https://aclanthology.org/2026.eamt-2.29/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>91</start>
<end>103</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Is a Picture Worth a Thousand Words? Exploration and Implementation Considerations for Visual Context in Translation Workflows
%A Guerra, Vera Senderowicz
%A Khrapunova, Olesia
%Y Shterionov, Dimitar
%Y Vanmassenhove, Eva
%Y De Sisto, Mirella
%Y Blain, Fred
%Y Pourmostafa Roshan Sharami, Javad
%Y Lepp, Lisa
%Y Manna, Chiara
%Y Rescigno, Argentina Anna
%Y Karakanta, Alina
%Y Rigouts Terryn, Ayla
%Y Lardelli, Manuel
%Y Resende, Natalia
%Y Murgolo, Elena
%Y Hackenbuchner, Janiça
%Y Zaretskaya, Anna
%Y Esplà-Gomis, Miquel
%Y Etchegoyhen, Thierry
%Y Gromann, Dagmar
%Y Bawden, Rachel
%Y Haddow, Barry
%Y Szoc, Sara
%Y Forcada, Mikel
%Y Moniz, Helena
%S Proceedings of the 26th Annual Conference of the European Association for Machine Translation (Volume 2)
%D 2026
%8 June
%I European Association for Machine Translation
%C Tilburg, The Netherlands
%@ 9789403901404
%F guerra-khrapunova-2026-picture
%X Vision-language models (VLMs) have the potential to enhance machine translation (MT) by leveraging visual context alongside text, yet their real utility for production workflows remains unclear. We conduct a unified, multi-condition evaluation of six leading VLMs—both open and proprietary—on two challenging benchmarks (CoMMuTE and CaMMT), targeting lexical and cultural disambiguation respectively, with a domain-style case study simulating technical documentation localization. Results show that model performance varies widely, and the benefit of relevant images does not necessarily transfer across use cases. Proprietary models are notably sensitive to irrelevant images while open-source models are generally more stable; incorrect or contradicting visuals, by contrast, degrade translation across all models. Taken together, these findings make rigorous evaluation a necessary precondition for production deployment: metric gains can mask real accuracy losses in technical domains, model sensitivity to irrelevant images should inform model selection, and reliable image–text matching is a hard requirement for any pipeline.
%U https://aclanthology.org/2026.eamt-2.29/
%P 91-103
Markdown (Informal)
[Is a Picture Worth a Thousand Words? Exploration and Implementation Considerations for Visual Context in Translation Workflows](https://aclanthology.org/2026.eamt-2.29/) (Guerra & Khrapunova, EAMT 2026)
ACL