@inproceedings{zhang-coltekin-2026-quantifying,
title = "Quantifying and Predicting Disagreement in Graded Human Ratings",
author = {Zhang, Leixin and
{\c{C}}{\"o}ltekin, {\c{C}}a{\u{g}}r{\i}},
editor = "Dudy, Shiran and
Abercrombie, Gavin and
Basile, Valerio and
Leonardelli, Elisa and
Frenda, Simona",
booktitle = "Proceedings of the the fifth edition of {NLP}erspectives",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.nlperspectives-1.4/",
doi = "10.63317/4qy8nuowzhpy",
pages = "33--43",
abstract = "It is increasingly recognized that humans do not always agree, and disagreement is inherent in many annotation tasks. However, not all items in a given task elicit the same level of opinion divergence. In this paper, we study the extent to which item-level annotation variation and variation structure can be captured from text features, focusing on inappropriate language detection, including offensive language, hate speech, and toxic language detection. We model annotation variation to assess whether the degree of annotation divergence can be predicted from item-level textual features. We also propose the Opposition Index, a metric that quantifies the extent of opposing stances among annotators based on their Likert ratings."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="zhang-coltekin-2026-quantifying">
<titleInfo>
<title>Quantifying and Predicting Disagreement in Graded Human Ratings</title>
</titleInfo>
<name type="personal">
<namePart type="given">Leixin</namePart>
<namePart type="family">Zhang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Çağrı</namePart>
<namePart type="family">Çöltekin</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the the fifth edition of NLPerspectives</title>
</titleInfo>
<name type="personal">
<namePart type="given">Shiran</namePart>
<namePart type="family">Dudy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gavin</namePart>
<namePart type="family">Abercrombie</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Valerio</namePart>
<namePart type="family">Basile</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elisa</namePart>
<namePart type="family">Leonardelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simona</namePart>
<namePart type="family">Frenda</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>It is increasingly recognized that humans do not always agree, and disagreement is inherent in many annotation tasks. However, not all items in a given task elicit the same level of opinion divergence. In this paper, we study the extent to which item-level annotation variation and variation structure can be captured from text features, focusing on inappropriate language detection, including offensive language, hate speech, and toxic language detection. We model annotation variation to assess whether the degree of annotation divergence can be predicted from item-level textual features. We also propose the Opposition Index, a metric that quantifies the extent of opposing stances among annotators based on their Likert ratings.</abstract>
<identifier type="citekey">zhang-coltekin-2026-quantifying</identifier>
<identifier type="doi">10.63317/4qy8nuowzhpy</identifier>
<location>
<url>https://aclanthology.org/2026.nlperspectives-1.4/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>33</start>
<end>43</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Quantifying and Predicting Disagreement in Graded Human Ratings
%A Zhang, Leixin
%A Çöltekin, Çağrı
%Y Dudy, Shiran
%Y Abercrombie, Gavin
%Y Basile, Valerio
%Y Leonardelli, Elisa
%Y Frenda, Simona
%S Proceedings of the the fifth edition of NLPerspectives
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F zhang-coltekin-2026-quantifying
%X It is increasingly recognized that humans do not always agree, and disagreement is inherent in many annotation tasks. However, not all items in a given task elicit the same level of opinion divergence. In this paper, we study the extent to which item-level annotation variation and variation structure can be captured from text features, focusing on inappropriate language detection, including offensive language, hate speech, and toxic language detection. We model annotation variation to assess whether the degree of annotation divergence can be predicted from item-level textual features. We also propose the Opposition Index, a metric that quantifies the extent of opposing stances among annotators based on their Likert ratings.
%R 10.63317/4qy8nuowzhpy
%U https://aclanthology.org/2026.nlperspectives-1.4/
%U https://doi.org/10.63317/4qy8nuowzhpy
%P 33-43
Markdown (Informal)
[Quantifying and Predicting Disagreement in Graded Human Ratings](https://aclanthology.org/2026.nlperspectives-1.4/) (Zhang & Çöltekin, NLPerspectives 2026)
ACL