@inproceedings{berger-hefetz-etal-2026-annotation,
title = "Annotation Matters: Resolving Cross-Corpus Performance Drops in {H}ebrew Offensive Language Detection",
author = "Berger Hefetz, Gili and
Shrem, Yossef Haim and
Vanetik, Natalia and
Liebeskind, Chaya",
editor = "Bagdon, Christopher and
Vishnubhotla, Krishnapriya and
Lindquist, Kristen A. and
Ungar, Lyle and
Klinger, Roman and
Mohammad, Saif M.",
booktitle = "Proceedings of Computational Affective Science ({CAS}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cas-1.10/",
doi = "10.63317/3fss6oc5mono",
pages = "116--124",
abstract = "Cross-dataset generalization remains a major challenge in offensive language detection, especially for culturally sensitive languages such as Hebrew. A large Hebrew dataset introduced in prior work (citation omitted for double-blind review) was annotated via a taxonomy-grounded, prompt-guided LLM protocol and achieved strong in-domain results. However, performance degraded sharply on two external Hebrew corpora. We investigate whether this degradation reflects domain shift or annotation shift, i.e., differences in how offensiveness is operationalized across datasets. Using the same prompt framework and a dual-LLM agreement procedure, we re-annotate both external corpora and quantify label divergence. We observe substantial mismatch between the original and new annotations, consistent with the view that offensiveness is not objective but depends on cultural context, discourse conventions, political framing, and the interpretation of irony. Evaluating models against the new labels yields markedly improved performance, and fine-tuning with the new external labels further improves results. Overall, our findings suggest that cross-dataset failure in affective NLP tasks may often be driven by annotation mismatch rather than domain adaptation limitations, highlighting the importance of annotation validity and culturally grounded labeling protocols."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="berger-hefetz-etal-2026-annotation">
<titleInfo>
<title>Annotation Matters: Resolving Cross-Corpus Performance Drops in Hebrew Offensive Language Detection</title>
</titleInfo>
<name type="personal">
<namePart type="given">Gili</namePart>
<namePart type="family">Berger Hefetz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yossef</namePart>
<namePart type="given">Haim</namePart>
<namePart type="family">Shrem</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Natalia</namePart>
<namePart type="family">Vanetik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chaya</namePart>
<namePart type="family">Liebeskind</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Computational Affective Science (CAS) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Bagdon</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Krishnapriya</namePart>
<namePart type="family">Vishnubhotla</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kristen</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Lindquist</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lyle</namePart>
<namePart type="family">Ungar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Roman</namePart>
<namePart type="family">Klinger</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saif</namePart>
<namePart type="given">M</namePart>
<namePart type="family">Mohammad</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Cross-dataset generalization remains a major challenge in offensive language detection, especially for culturally sensitive languages such as Hebrew. A large Hebrew dataset introduced in prior work (citation omitted for double-blind review) was annotated via a taxonomy-grounded, prompt-guided LLM protocol and achieved strong in-domain results. However, performance degraded sharply on two external Hebrew corpora. We investigate whether this degradation reflects domain shift or annotation shift, i.e., differences in how offensiveness is operationalized across datasets. Using the same prompt framework and a dual-LLM agreement procedure, we re-annotate both external corpora and quantify label divergence. We observe substantial mismatch between the original and new annotations, consistent with the view that offensiveness is not objective but depends on cultural context, discourse conventions, political framing, and the interpretation of irony. Evaluating models against the new labels yields markedly improved performance, and fine-tuning with the new external labels further improves results. Overall, our findings suggest that cross-dataset failure in affective NLP tasks may often be driven by annotation mismatch rather than domain adaptation limitations, highlighting the importance of annotation validity and culturally grounded labeling protocols.</abstract>
<identifier type="citekey">berger-hefetz-etal-2026-annotation</identifier>
<identifier type="doi">10.63317/3fss6oc5mono</identifier>
<location>
<url>https://aclanthology.org/2026.cas-1.10/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>116</start>
<end>124</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Annotation Matters: Resolving Cross-Corpus Performance Drops in Hebrew Offensive Language Detection
%A Berger Hefetz, Gili
%A Shrem, Yossef Haim
%A Vanetik, Natalia
%A Liebeskind, Chaya
%Y Bagdon, Christopher
%Y Vishnubhotla, Krishnapriya
%Y Lindquist, Kristen A.
%Y Ungar, Lyle
%Y Klinger, Roman
%Y Mohammad, Saif M.
%S Proceedings of Computational Affective Science (CAS) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F berger-hefetz-etal-2026-annotation
%X Cross-dataset generalization remains a major challenge in offensive language detection, especially for culturally sensitive languages such as Hebrew. A large Hebrew dataset introduced in prior work (citation omitted for double-blind review) was annotated via a taxonomy-grounded, prompt-guided LLM protocol and achieved strong in-domain results. However, performance degraded sharply on two external Hebrew corpora. We investigate whether this degradation reflects domain shift or annotation shift, i.e., differences in how offensiveness is operationalized across datasets. Using the same prompt framework and a dual-LLM agreement procedure, we re-annotate both external corpora and quantify label divergence. We observe substantial mismatch between the original and new annotations, consistent with the view that offensiveness is not objective but depends on cultural context, discourse conventions, political framing, and the interpretation of irony. Evaluating models against the new labels yields markedly improved performance, and fine-tuning with the new external labels further improves results. Overall, our findings suggest that cross-dataset failure in affective NLP tasks may often be driven by annotation mismatch rather than domain adaptation limitations, highlighting the importance of annotation validity and culturally grounded labeling protocols.
%R 10.63317/3fss6oc5mono
%U https://aclanthology.org/2026.cas-1.10/
%U https://doi.org/10.63317/3fss6oc5mono
%P 116-124
Markdown (Informal)
[Annotation Matters: Resolving Cross-Corpus Performance Drops in Hebrew Offensive Language Detection](https://aclanthology.org/2026.cas-1.10/) (Berger Hefetz et al., CAS 2026)
ACL