@inproceedings{said-villaneau-2026-multi,
title = "Multi-Source Emotion Annotation in Children{'}s Language: When {LLM} Consensus Diverges from Human Judgment",
author = "Said, Farida and
Villaneau, Jeanne",
editor = "Bagdon, Christopher and
Vishnubhotla, Krishnapriya and
Lindquist, Kristen A. and
Ungar, Lyle and
Klinger, Roman and
Mohammad, Saif M.",
booktitle = "Proceedings of Computational Affective Science ({CAS}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cas-1.11/",
doi = "10.63317/39tk69v6ca4p",
pages = "125--135",
abstract = "Automated emotion annotation increasingly relies on inter-LLM agreement as a proxy for label quality. We test this assumption on 2,106 clause-level segments from interviews with French-speaking children (ages 6-11) about parental roles, a setting where affect is often implicit rather than lexically explicit. Using a 500-segment expert gold standard, we show that internal consensus can be seriously misleading: Dawid-Skene, a probabilistic label aggregation method, estimates GPT-5.2 valence accuracy at 90.7{\%}, whereas evaluation against human gold yields 71.0{\%}, revealing substantial overestimation driven by shared neutralization bias. Conversely, Dawid-Skene underestimates Claude Sonnet 4, reversing model ranking. Majority Vote, Dawid-Skene, and MACE produce near-identical consensus labels, suggesting that the main source of error lies in shared annotator bias rather than in the aggregation rule itself. We release the expert gold subset and the probabilistic corpus to support future work. Our results show that high inter-LLM agreement cannot replace external human validation for affect annotation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="said-villaneau-2026-multi">
<titleInfo>
<title>Multi-Source Emotion Annotation in Children’s Language: When LLM Consensus Diverges from Human Judgment</title>
</titleInfo>
<name type="personal">
<namePart type="given">Farida</namePart>
<namePart type="family">Said</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jeanne</namePart>
<namePart type="family">Villaneau</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Computational Affective Science (CAS) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Bagdon</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Krishnapriya</namePart>
<namePart type="family">Vishnubhotla</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kristen</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Lindquist</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lyle</namePart>
<namePart type="family">Ungar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Roman</namePart>
<namePart type="family">Klinger</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saif</namePart>
<namePart type="given">M</namePart>
<namePart type="family">Mohammad</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Automated emotion annotation increasingly relies on inter-LLM agreement as a proxy for label quality. We test this assumption on 2,106 clause-level segments from interviews with French-speaking children (ages 6-11) about parental roles, a setting where affect is often implicit rather than lexically explicit. Using a 500-segment expert gold standard, we show that internal consensus can be seriously misleading: Dawid-Skene, a probabilistic label aggregation method, estimates GPT-5.2 valence accuracy at 90.7%, whereas evaluation against human gold yields 71.0%, revealing substantial overestimation driven by shared neutralization bias. Conversely, Dawid-Skene underestimates Claude Sonnet 4, reversing model ranking. Majority Vote, Dawid-Skene, and MACE produce near-identical consensus labels, suggesting that the main source of error lies in shared annotator bias rather than in the aggregation rule itself. We release the expert gold subset and the probabilistic corpus to support future work. Our results show that high inter-LLM agreement cannot replace external human validation for affect annotation.</abstract>
<identifier type="citekey">said-villaneau-2026-multi</identifier>
<identifier type="doi">10.63317/39tk69v6ca4p</identifier>
<location>
<url>https://aclanthology.org/2026.cas-1.11/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>125</start>
<end>135</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Multi-Source Emotion Annotation in Children’s Language: When LLM Consensus Diverges from Human Judgment
%A Said, Farida
%A Villaneau, Jeanne
%Y Bagdon, Christopher
%Y Vishnubhotla, Krishnapriya
%Y Lindquist, Kristen A.
%Y Ungar, Lyle
%Y Klinger, Roman
%Y Mohammad, Saif M.
%S Proceedings of Computational Affective Science (CAS) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F said-villaneau-2026-multi
%X Automated emotion annotation increasingly relies on inter-LLM agreement as a proxy for label quality. We test this assumption on 2,106 clause-level segments from interviews with French-speaking children (ages 6-11) about parental roles, a setting where affect is often implicit rather than lexically explicit. Using a 500-segment expert gold standard, we show that internal consensus can be seriously misleading: Dawid-Skene, a probabilistic label aggregation method, estimates GPT-5.2 valence accuracy at 90.7%, whereas evaluation against human gold yields 71.0%, revealing substantial overestimation driven by shared neutralization bias. Conversely, Dawid-Skene underestimates Claude Sonnet 4, reversing model ranking. Majority Vote, Dawid-Skene, and MACE produce near-identical consensus labels, suggesting that the main source of error lies in shared annotator bias rather than in the aggregation rule itself. We release the expert gold subset and the probabilistic corpus to support future work. Our results show that high inter-LLM agreement cannot replace external human validation for affect annotation.
%R 10.63317/39tk69v6ca4p
%U https://aclanthology.org/2026.cas-1.11/
%U https://doi.org/10.63317/39tk69v6ca4p
%P 125-135
Markdown (Informal)
[Multi-Source Emotion Annotation in Children’s Language: When LLM Consensus Diverges from Human Judgment](https://aclanthology.org/2026.cas-1.11/) (Said & Villaneau, CAS 2026)
ACL