@inproceedings{tao-etal-2026-disentangling,
title = "Disentangling Annotator Skill from Verifier Strictness in Cross-Verified Dialogue Annotation",
author = "Tao, Zihao and
Prado, John A. and
LaManna, Ignazio Steven and
Puterbaugh, Ryan and
Datta, Mim and
Hirschberg, Julia",
editor = "Choi, Jinho D. and
Chen, Yun-Nung and
Funakoshi, Kotaro and
Emami, Ali",
booktitle = "Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue",
month = aug,
year = "2026",
address = "Atlanta, Georgia, USA",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.sigdial-1.29/",
pages = "414--422",
abstract = "Some dialogue corpus projects use a verify-after-annotation workflow: a second team member reviews a submitted file and records corrections. The resulting correction count mixes two signals, annotator accuracy and verifier strictness. We separate these signals for RASwDA, an audio-anchored re-alignment of 1,045 Switchboard file sides (105,005 corrections across 977 change logs) produced by five team members in 2024-2025. Verification is crossed: each of three identified annotators was checked by four different verifiers, with overlap in both directions. A cross-classified mixed-effects model on per-file corrections-per-interval assigns 10.8{\%} of the variance to annotator identity, while the verifier random effect collapses to zero (singular fit, stable across seven leave-one-out and response-choice refits). Thus, for this boundary-realignment task, we find no detectable verifier identity effect once annotator identity, batch, and file length are controlled. Boundary placement, not label selection, accounts for 58.5{\%} of corrections corpus-wide. This helps explain why verifier-specific strictness has little room to appear: timestamp adjustments are anchored in the audio, while DA-label changes account for only 3.0{\%} of corrections. We release the action-typed change logs so other projects can run the same annotator-verifier decomposition on their own verification data."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="tao-etal-2026-disentangling">
<titleInfo>
<title>Disentangling Annotator Skill from Verifier Strictness in Cross-Verified Dialogue Annotation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Zihao</namePart>
<namePart type="family">Tao</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">John</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Prado</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ignazio</namePart>
<namePart type="given">Steven</namePart>
<namePart type="family">LaManna</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ryan</namePart>
<namePart type="family">Puterbaugh</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mim</namePart>
<namePart type="family">Datta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Julia</namePart>
<namePart type="family">Hirschberg</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-08</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jinho</namePart>
<namePart type="given">D</namePart>
<namePart type="family">Choi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yun-Nung</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kotaro</namePart>
<namePart type="family">Funakoshi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ali</namePart>
<namePart type="family">Emami</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Atlanta, Georgia, USA</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Some dialogue corpus projects use a verify-after-annotation workflow: a second team member reviews a submitted file and records corrections. The resulting correction count mixes two signals, annotator accuracy and verifier strictness. We separate these signals for RASwDA, an audio-anchored re-alignment of 1,045 Switchboard file sides (105,005 corrections across 977 change logs) produced by five team members in 2024-2025. Verification is crossed: each of three identified annotators was checked by four different verifiers, with overlap in both directions. A cross-classified mixed-effects model on per-file corrections-per-interval assigns 10.8% of the variance to annotator identity, while the verifier random effect collapses to zero (singular fit, stable across seven leave-one-out and response-choice refits). Thus, for this boundary-realignment task, we find no detectable verifier identity effect once annotator identity, batch, and file length are controlled. Boundary placement, not label selection, accounts for 58.5% of corrections corpus-wide. This helps explain why verifier-specific strictness has little room to appear: timestamp adjustments are anchored in the audio, while DA-label changes account for only 3.0% of corrections. We release the action-typed change logs so other projects can run the same annotator-verifier decomposition on their own verification data.</abstract>
<identifier type="citekey">tao-etal-2026-disentangling</identifier>
<location>
<url>https://aclanthology.org/2026.sigdial-1.29/</url>
</location>
<part>
<date>2026-08</date>
<extent unit="page">
<start>414</start>
<end>422</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Disentangling Annotator Skill from Verifier Strictness in Cross-Verified Dialogue Annotation
%A Tao, Zihao
%A Prado, John A.
%A LaManna, Ignazio Steven
%A Puterbaugh, Ryan
%A Datta, Mim
%A Hirschberg, Julia
%Y Choi, Jinho D.
%Y Chen, Yun-Nung
%Y Funakoshi, Kotaro
%Y Emami, Ali
%S Proceedings of the 27th Annual Meeting of the Special Interest Group on Discourse and Dialogue
%D 2026
%8 August
%I Association for Computational Linguistics
%C Atlanta, Georgia, USA
%F tao-etal-2026-disentangling
%X Some dialogue corpus projects use a verify-after-annotation workflow: a second team member reviews a submitted file and records corrections. The resulting correction count mixes two signals, annotator accuracy and verifier strictness. We separate these signals for RASwDA, an audio-anchored re-alignment of 1,045 Switchboard file sides (105,005 corrections across 977 change logs) produced by five team members in 2024-2025. Verification is crossed: each of three identified annotators was checked by four different verifiers, with overlap in both directions. A cross-classified mixed-effects model on per-file corrections-per-interval assigns 10.8% of the variance to annotator identity, while the verifier random effect collapses to zero (singular fit, stable across seven leave-one-out and response-choice refits). Thus, for this boundary-realignment task, we find no detectable verifier identity effect once annotator identity, batch, and file length are controlled. Boundary placement, not label selection, accounts for 58.5% of corrections corpus-wide. This helps explain why verifier-specific strictness has little room to appear: timestamp adjustments are anchored in the audio, while DA-label changes account for only 3.0% of corrections. We release the action-typed change logs so other projects can run the same annotator-verifier decomposition on their own verification data.
%U https://aclanthology.org/2026.sigdial-1.29/
%P 414-422
Markdown (Informal)
[Disentangling Annotator Skill from Verifier Strictness in Cross-Verified Dialogue Annotation](https://aclanthology.org/2026.sigdial-1.29/) (Tao et al., SIGDIAL 2026)
ACL