@inproceedings{said-villaneau-2026-multi,
title = "Multi-Source Emotion Annotation in Children{'}s Language: When {LLM} Consensus Diverges from Human Judgment",
author = "Said, Farida and
Villaneau, Jeanne",
editor = "Bagdon, Christopher and
Vishnubhotla, Krishnapriya and
Lindquist, Kristen A. and
Ungar, Lyle and
Klinger, Roman and
Mohammad, Saif M.",
booktitle = "Proceedings of Computational Affective Science ({CAS}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://preview.aclanthology.org/test-year-match/2026.cas-1.11/",
doi = "10.63317/39tk69v6ca4p",
pages = "125--135",
abstract = "Automated emotion annotation increasingly relies on inter-LLM agreement as a proxy for label quality. We test this assumption on 2,106 clause-level segments from interviews with French-speaking children (ages 6-11) about parental roles, a setting where affect is often implicit rather than lexically explicit. Using a 500-segment expert gold standard, we show that internal consensus can be seriously misleading: Dawid-Skene, a probabilistic label aggregation method, estimates GPT-5.2 valence accuracy at 90.7{\%}, whereas evaluation against human gold yields 71.0{\%}, revealing substantial overestimation driven by shared neutralization bias. Conversely, Dawid-Skene underestimates Claude Sonnet 4, reversing model ranking. Majority Vote, Dawid-Skene, and MACE produce near-identical consensus labels, suggesting that the main source of error lies in shared annotator bias rather than in the aggregation rule itself. We release the expert gold subset and the probabilistic corpus to support future work. Our results show that high inter-LLM agreement cannot replace external human validation for affect annotation."
}Markdown (Informal)
[Multi-Source Emotion Annotation in Children’s Language: When LLM Consensus Diverges from Human Judgment](https://preview.aclanthology.org/test-year-match/2026.cas-1.11/) (Said & Villaneau, CAS 2026)
ACL