@inproceedings{rezapoor-etal-2026-predicting,
title = "Predicting Gaze Location without Camera or Eye-Tracker",
author = "Rezapoor, Saman and
Shirali-Shahreza, Sajad and
Penn, Gerald",
editor = {Acart{\"u}rk, Cengiz and
Can, Burcu and
Nasir, Jamal and
{\c{C}}{\"o}ltekin, {\c{C}}a{\u{g}}r{\i}},
booktitle = "Proceedings fo the Second International Workshop on Eye-Tracking Resources and Evaluation for Human-Aligned {NLP}",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELDA",
url = "https://aclanthology.org/2026.gaze4nlp-1.9/",
doi = "10.63317/2hpa22f63k2t",
pages = "58--63",
abstract = "The task of identifying the location that a user looks at, commonly known as gaze estimation, has various HCI and NLP applications. Traditional gaze estimation methods use special hardware such as eye-trackers or ordinary cameras such as webcams to perform this. However, they are not applicable to the majority of web users either because the user does not have them or does not want to use them due to privacy reasons. In this paper, we propose the idea of using multimodal LLMs to analyze the content of the user{'}s screen along with mouse location to estimate the gaze location. It primarily uses the results of studies that extract common reading patterns such as the F-pattern and Z-pattern. Our experimental results on The Eye Of The Typer (EOTT) dataset provide promising results for estimating gaze location."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="rezapoor-etal-2026-predicting">
<titleInfo>
<title>Predicting Gaze Location without Camera or Eye-Tracker</title>
</titleInfo>
<name type="personal">
<namePart type="given">Saman</namePart>
<namePart type="family">Rezapoor</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sajad</namePart>
<namePart type="family">Shirali-Shahreza</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gerald</namePart>
<namePart type="family">Penn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings fo the Second International Workshop on Eye-Tracking Resources and Evaluation for Human-Aligned NLP</title>
</titleInfo>
<name type="personal">
<namePart type="given">Cengiz</namePart>
<namePart type="family">Acartürk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Burcu</namePart>
<namePart type="family">Can</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jamal</namePart>
<namePart type="family">Nasir</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Çağrı</namePart>
<namePart type="family">Çöltekin</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELDA</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The task of identifying the location that a user looks at, commonly known as gaze estimation, has various HCI and NLP applications. Traditional gaze estimation methods use special hardware such as eye-trackers or ordinary cameras such as webcams to perform this. However, they are not applicable to the majority of web users either because the user does not have them or does not want to use them due to privacy reasons. In this paper, we propose the idea of using multimodal LLMs to analyze the content of the user’s screen along with mouse location to estimate the gaze location. It primarily uses the results of studies that extract common reading patterns such as the F-pattern and Z-pattern. Our experimental results on The Eye Of The Typer (EOTT) dataset provide promising results for estimating gaze location.</abstract>
<identifier type="citekey">rezapoor-etal-2026-predicting</identifier>
<identifier type="doi">10.63317/2hpa22f63k2t</identifier>
<location>
<url>https://aclanthology.org/2026.gaze4nlp-1.9/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>58</start>
<end>63</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Predicting Gaze Location without Camera or Eye-Tracker
%A Rezapoor, Saman
%A Shirali-Shahreza, Sajad
%A Penn, Gerald
%Y Acartürk, Cengiz
%Y Can, Burcu
%Y Nasir, Jamal
%Y Çöltekin, Çağrı
%S Proceedings fo the Second International Workshop on Eye-Tracking Resources and Evaluation for Human-Aligned NLP
%D 2026
%8 May
%I ELDA
%C Palma de Mallorca, Spain
%F rezapoor-etal-2026-predicting
%X The task of identifying the location that a user looks at, commonly known as gaze estimation, has various HCI and NLP applications. Traditional gaze estimation methods use special hardware such as eye-trackers or ordinary cameras such as webcams to perform this. However, they are not applicable to the majority of web users either because the user does not have them or does not want to use them due to privacy reasons. In this paper, we propose the idea of using multimodal LLMs to analyze the content of the user’s screen along with mouse location to estimate the gaze location. It primarily uses the results of studies that extract common reading patterns such as the F-pattern and Z-pattern. Our experimental results on The Eye Of The Typer (EOTT) dataset provide promising results for estimating gaze location.
%R 10.63317/2hpa22f63k2t
%U https://aclanthology.org/2026.gaze4nlp-1.9/
%U https://doi.org/10.63317/2hpa22f63k2t
%P 58-63
Markdown (Informal)
[Predicting Gaze Location without Camera or Eye-Tracker](https://aclanthology.org/2026.gaze4nlp-1.9/) (Rezapoor et al., Gaze4NLP 2026)
ACL
- Saman Rezapoor, Sajad Shirali-Shahreza, and Gerald Penn. 2026. Predicting Gaze Location without Camera or Eye-Tracker. In Proceedings fo the Second International Workshop on Eye-Tracking Resources and Evaluation for Human-Aligned NLP, pages 58–63, Palma de Mallorca, Spain. ELDA.