@inproceedings{jandial-etal-2025-fine,
title = "On the Fine-Grained Planning Abilities of {VLM} Web Agents",
author = "Jandial, Surgan and
Wang, Yinong Oliver and
Bajcsy, Andrea and
De la Torre, Fernando",
editor = "Christodoulopoulos, Christos and
Chakraborty, Tanmoy and
Rose, Carolyn and
Peng, Violet",
booktitle = "Findings of the Association for Computational Linguistics: EMNLP 2025",
month = nov,
year = "2025",
address = "Suzhou, China",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2025.findings-emnlp.1382/",
doi = "10.18653/v1/2025.findings-emnlp.1382",
pages = "25347--25380",
ISBN = "979-8-89176-335-7",
abstract = "Vision-Language Models (VLMs) have shown promise as web agents, yet their planning{---}the ability to devise strategies or action sequences to complete tasks{---}remains understudied. While prior works focus on VLM{'}s perception and overall success rates (i.e., goal completion), fine-grained investigation of their planning has been overlooked. To address this gap, we examine VLMs' capability to (1) understand temporal relationships within web contexts, and (2) assess plans of actions across diverse scenarios. We design four simple yet effective tests to delve into these nuanced aspects around planning. Our results across nineteen VLMs reveal that these models exhibit limited performance in the aforementioned skills and are not reliable to function as web agents. To facilitate future work, we release our planning evaluations and data, providing a foundation for advancing the future research in this area."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="jandial-etal-2025-fine">
<titleInfo>
<title>On the Fine-Grained Planning Abilities of VLM Web Agents</title>
</titleInfo>
<name type="personal">
<namePart type="given">Surgan</namePart>
<namePart type="family">Jandial</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yinong</namePart>
<namePart type="given">Oliver</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andrea</namePart>
<namePart type="family">Bajcsy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fernando</namePart>
<namePart type="family">De la Torre</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2025-11</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Findings of the Association for Computational Linguistics: EMNLP 2025</title>
</titleInfo>
<name type="personal">
<namePart type="given">Christos</namePart>
<namePart type="family">Christodoulopoulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tanmoy</namePart>
<namePart type="family">Chakraborty</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Carolyn</namePart>
<namePart type="family">Rose</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Violet</namePart>
<namePart type="family">Peng</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Suzhou, China</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-89176-335-7</identifier>
</relatedItem>
<abstract>Vision-Language Models (VLMs) have shown promise as web agents, yet their planning—the ability to devise strategies or action sequences to complete tasks—remains understudied. While prior works focus on VLM’s perception and overall success rates (i.e., goal completion), fine-grained investigation of their planning has been overlooked. To address this gap, we examine VLMs’ capability to (1) understand temporal relationships within web contexts, and (2) assess plans of actions across diverse scenarios. We design four simple yet effective tests to delve into these nuanced aspects around planning. Our results across nineteen VLMs reveal that these models exhibit limited performance in the aforementioned skills and are not reliable to function as web agents. To facilitate future work, we release our planning evaluations and data, providing a foundation for advancing the future research in this area.</abstract>
<identifier type="citekey">jandial-etal-2025-fine</identifier>
<identifier type="doi">10.18653/v1/2025.findings-emnlp.1382</identifier>
<location>
<url>https://aclanthology.org/2025.findings-emnlp.1382/</url>
</location>
<part>
<date>2025-11</date>
<extent unit="page">
<start>25347</start>
<end>25380</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T On the Fine-Grained Planning Abilities of VLM Web Agents
%A Jandial, Surgan
%A Wang, Yinong Oliver
%A Bajcsy, Andrea
%A De la Torre, Fernando
%Y Christodoulopoulos, Christos
%Y Chakraborty, Tanmoy
%Y Rose, Carolyn
%Y Peng, Violet
%S Findings of the Association for Computational Linguistics: EMNLP 2025
%D 2025
%8 November
%I Association for Computational Linguistics
%C Suzhou, China
%@ 979-8-89176-335-7
%F jandial-etal-2025-fine
%X Vision-Language Models (VLMs) have shown promise as web agents, yet their planning—the ability to devise strategies or action sequences to complete tasks—remains understudied. While prior works focus on VLM’s perception and overall success rates (i.e., goal completion), fine-grained investigation of their planning has been overlooked. To address this gap, we examine VLMs’ capability to (1) understand temporal relationships within web contexts, and (2) assess plans of actions across diverse scenarios. We design four simple yet effective tests to delve into these nuanced aspects around planning. Our results across nineteen VLMs reveal that these models exhibit limited performance in the aforementioned skills and are not reliable to function as web agents. To facilitate future work, we release our planning evaluations and data, providing a foundation for advancing the future research in this area.
%R 10.18653/v1/2025.findings-emnlp.1382
%U https://aclanthology.org/2025.findings-emnlp.1382/
%U https://doi.org/10.18653/v1/2025.findings-emnlp.1382
%P 25347-25380
Markdown (Informal)
[On the Fine-Grained Planning Abilities of VLM Web Agents](https://aclanthology.org/2025.findings-emnlp.1382/) (Jandial et al., Findings 2025)
ACL
- Surgan Jandial, Yinong Oliver Wang, Andrea Bajcsy, and Fernando De la Torre. 2025. On the Fine-Grained Planning Abilities of VLM Web Agents. In Findings of the Association for Computational Linguistics: EMNLP 2025, pages 25347–25380, Suzhou, China. Association for Computational Linguistics.