@article{dos-santos-leal-2026-large,
title = "Can a Large Language Model Replace Humans at Rating Lexical Semantic Relations Strength?",
author = "dos Santos, Andr{\'e} Fernandes and
Leal, Jos{\'e} Paulo",
journal = "Computational Linguistics",
volume = "52",
number = "2",
month = jun,
year = "2026",
address = "Cambridge, MA",
publisher = "MIT Press",
url = "https://aclanthology.org/2026.cl-2.7/",
doi = "10.1162/coli.a.587",
pages = "677--723",
abstract = "This article investigates the ability of large language models (LLMs) to evaluate semantic relations between word pairs by examining their alignment with human-generated semantic ratings. Semantic relations represent the degree of connection (e.g., relatedness or similarity) between linguistic elements and are traditionally validated against human-annotated datasets. Due to the challenges of building such datasets and recent progress in LLMs' capacity to model human-like understanding, we explore whether LLMs can serve as reliable substitutes for traditional human ratings. We conducted experiments using multiple LLMs from OpenAI, Google, Mistral, and Anthropic, evaluating their performance across diverse English and Portuguese semantic relations datasets. We included in the analysis PAP900, a recently published dataset of semantic relations in Portuguese, to examine the influence of prior exposure to the dataset on LLM training. The results show that the LLM predictions correlate strongly with human ratings. The findings reveal the potential of LLMs to supplement or replace traditional semantic measure algorithms and crowd-sourced human annotations in semantic tasks."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="dos-santos-leal-2026-large">
<titleInfo>
<title>Can a Large Language Model Replace Humans at Rating Lexical Semantic Relations Strength?</title>
</titleInfo>
<name type="personal">
<namePart type="given">André</namePart>
<namePart type="given">Fernandes</namePart>
<namePart type="family">dos Santos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">José</namePart>
<namePart type="given">Paulo</namePart>
<namePart type="family">Leal</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<genre authority="bibutilsgt">journal article</genre>
<relatedItem type="host">
<titleInfo>
<title>Computational Linguistics</title>
</titleInfo>
<originInfo>
<issuance>continuing</issuance>
<publisher>MIT Press</publisher>
<place>
<placeTerm type="text">Cambridge, MA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">periodical</genre>
<genre authority="bibutilsgt">academic journal</genre>
</relatedItem>
<abstract>This article investigates the ability of large language models (LLMs) to evaluate semantic relations between word pairs by examining their alignment with human-generated semantic ratings. Semantic relations represent the degree of connection (e.g., relatedness or similarity) between linguistic elements and are traditionally validated against human-annotated datasets. Due to the challenges of building such datasets and recent progress in LLMs’ capacity to model human-like understanding, we explore whether LLMs can serve as reliable substitutes for traditional human ratings. We conducted experiments using multiple LLMs from OpenAI, Google, Mistral, and Anthropic, evaluating their performance across diverse English and Portuguese semantic relations datasets. We included in the analysis PAP900, a recently published dataset of semantic relations in Portuguese, to examine the influence of prior exposure to the dataset on LLM training. The results show that the LLM predictions correlate strongly with human ratings. The findings reveal the potential of LLMs to supplement or replace traditional semantic measure algorithms and crowd-sourced human annotations in semantic tasks.</abstract>
<identifier type="citekey">dos-santos-leal-2026-large</identifier>
<identifier type="doi">10.1162/coli.a.587</identifier>
<location>
<url>https://aclanthology.org/2026.cl-2.7/</url>
</location>
<part>
<date>2026-06</date>
<detail type="volume"><number>52</number></detail>
<detail type="issue"><number>2</number></detail>
<extent unit="page">
<start>677</start>
<end>723</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Journal Article
%T Can a Large Language Model Replace Humans at Rating Lexical Semantic Relations Strength?
%A dos Santos, André Fernandes
%A Leal, José Paulo
%J Computational Linguistics
%D 2026
%8 June
%V 52
%N 2
%I MIT Press
%C Cambridge, MA
%F dos-santos-leal-2026-large
%X This article investigates the ability of large language models (LLMs) to evaluate semantic relations between word pairs by examining their alignment with human-generated semantic ratings. Semantic relations represent the degree of connection (e.g., relatedness or similarity) between linguistic elements and are traditionally validated against human-annotated datasets. Due to the challenges of building such datasets and recent progress in LLMs’ capacity to model human-like understanding, we explore whether LLMs can serve as reliable substitutes for traditional human ratings. We conducted experiments using multiple LLMs from OpenAI, Google, Mistral, and Anthropic, evaluating their performance across diverse English and Portuguese semantic relations datasets. We included in the analysis PAP900, a recently published dataset of semantic relations in Portuguese, to examine the influence of prior exposure to the dataset on LLM training. The results show that the LLM predictions correlate strongly with human ratings. The findings reveal the potential of LLMs to supplement or replace traditional semantic measure algorithms and crowd-sourced human annotations in semantic tasks.
%R 10.1162/coli.a.587
%U https://aclanthology.org/2026.cl-2.7/
%U https://doi.org/10.1162/coli.a.587
%P 677-723
Markdown (Informal)
[Can a Large Language Model Replace Humans at Rating Lexical Semantic Relations Strength?](https://aclanthology.org/2026.cl-2.7/) (dos Santos & Leal, CL 2026)
ACL