@inproceedings{abayomi-etal-2026-geobenchmark,
title = "{G}eo{B}enchmark: Probing Large Language Models for Geo-Spatial Knowledge",
author = "Abayomi, Ayomide and
Moreno, Jose G. and
Radouane, Karim and
Tamine, Lynda",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.417/",
doi = "10.63317/26pwz3584735",
pages = "5335--5348",
abstract = "Large Language Models (LLMs) demonstrate strong factual recall of general-purpose knowledge but struggle with grounded geospatial knowledge. To measure and help probe LLMs for spatial knowledge, we present GeoBenchmark, a benchmark for evaluating geographic commonsense along three core spatial relations: direction, distance, and topology. Using data extracted from YAGO2geo and Ordnance Survey ward geometries, spatial relations were formalized as structured triplets and systematically transformed into balanced binary (Yes/No) and Multiple-Choice (MCQ) question-answer pairs. Besides, we consider atomic and composite questions based on the number of spatial relations involved. The resulting dataset comprises 26k binary and 13k MCQ samples, uniformly distributed across atomic, binary, and ternary relation levels. We establish baselines with LLaMA-8B and Mistral-7B under zero-shot prompting, achieving 52-63{\%} accuracy on atomic questions but below 35{\%} on ternary relations, which exposes the models' limited compositional spatial understanding and strong option bias. GeoBenchmark provides a comprehensive, reproducible resource for probing and advancing LLMs' geographic commonsense, paving the way for future research in spatial and geographic probing of LLMs as well as knowledge editing."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="abayomi-etal-2026-geobenchmark">
<titleInfo>
<title>GeoBenchmark: Probing Large Language Models for Geo-Spatial Knowledge</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ayomide</namePart>
<namePart type="family">Abayomi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jose</namePart>
<namePart type="given">G</namePart>
<namePart type="family">Moreno</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Karim</namePart>
<namePart type="family">Radouane</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lynda</namePart>
<namePart type="family">Tamine</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Large Language Models (LLMs) demonstrate strong factual recall of general-purpose knowledge but struggle with grounded geospatial knowledge. To measure and help probe LLMs for spatial knowledge, we present GeoBenchmark, a benchmark for evaluating geographic commonsense along three core spatial relations: direction, distance, and topology. Using data extracted from YAGO2geo and Ordnance Survey ward geometries, spatial relations were formalized as structured triplets and systematically transformed into balanced binary (Yes/No) and Multiple-Choice (MCQ) question-answer pairs. Besides, we consider atomic and composite questions based on the number of spatial relations involved. The resulting dataset comprises 26k binary and 13k MCQ samples, uniformly distributed across atomic, binary, and ternary relation levels. We establish baselines with LLaMA-8B and Mistral-7B under zero-shot prompting, achieving 52-63% accuracy on atomic questions but below 35% on ternary relations, which exposes the models’ limited compositional spatial understanding and strong option bias. GeoBenchmark provides a comprehensive, reproducible resource for probing and advancing LLMs’ geographic commonsense, paving the way for future research in spatial and geographic probing of LLMs as well as knowledge editing.</abstract>
<identifier type="citekey">abayomi-etal-2026-geobenchmark</identifier>
<identifier type="doi">10.63317/26pwz3584735</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.417/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>5335</start>
<end>5348</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T GeoBenchmark: Probing Large Language Models for Geo-Spatial Knowledge
%A Abayomi, Ayomide
%A Moreno, Jose G.
%A Radouane, Karim
%A Tamine, Lynda
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F abayomi-etal-2026-geobenchmark
%X Large Language Models (LLMs) demonstrate strong factual recall of general-purpose knowledge but struggle with grounded geospatial knowledge. To measure and help probe LLMs for spatial knowledge, we present GeoBenchmark, a benchmark for evaluating geographic commonsense along three core spatial relations: direction, distance, and topology. Using data extracted from YAGO2geo and Ordnance Survey ward geometries, spatial relations were formalized as structured triplets and systematically transformed into balanced binary (Yes/No) and Multiple-Choice (MCQ) question-answer pairs. Besides, we consider atomic and composite questions based on the number of spatial relations involved. The resulting dataset comprises 26k binary and 13k MCQ samples, uniformly distributed across atomic, binary, and ternary relation levels. We establish baselines with LLaMA-8B and Mistral-7B under zero-shot prompting, achieving 52-63% accuracy on atomic questions but below 35% on ternary relations, which exposes the models’ limited compositional spatial understanding and strong option bias. GeoBenchmark provides a comprehensive, reproducible resource for probing and advancing LLMs’ geographic commonsense, paving the way for future research in spatial and geographic probing of LLMs as well as knowledge editing.
%R 10.63317/26pwz3584735
%U https://aclanthology.org/2026.lrec-1.417/
%U https://doi.org/10.63317/26pwz3584735
%P 5335-5348
Markdown (Informal)
[GeoBenchmark: Probing Large Language Models for Geo-Spatial Knowledge](https://aclanthology.org/2026.lrec-1.417/) (Abayomi et al., LREC 2026)
ACL