@inproceedings{coats-2026-md,
title = "{MD}{\_}{NLP}: Reconstructing an {A}ustralian {E}nglish Heritage Dialect Corpus from the {M}itchell-Delbridge Recordings through {LLM}-Assisted Speaker Attribution",
author = "Coats, Steven",
editor = "Anastasopoulos, Antonis and
Markantonatou, Stella and
Ralli, Angela and
Zampieri, Marcos and
Bompolas, Stavros and
Stamou, Vivian",
booktitle = "Proceedings of the First Workshop on Dialects in {NLP} {---} A Resource Perspective",
month = may,
year = "2026",
address = "Palma de Mallorca",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.dialres-1.3/",
doi = "10.63317/3iga3zzsh92p",
pages = "24--32",
abstract = "We present MD{\_}NLP, a discourse-annotated and georeferenced corpus derived from the Mitchell{--}Delbridge (MD) recordings, a foundational archive of mid-20th-century Australian English. The corpus comprises word-aligned narrative recordings from 7,735 secondary school pupils across 327 locations, enriched with structured sociodemographic metadata (e.g., sex, birthplace, parental background) and geocoded institutional coordinates. The narratives were reconstructed from archival audio using an integrated pipeline combining WhisperX-based automatic speech recognition, neural speaker diarization, LLM-assisted discourse-role correction, and Montreal Forced Aligner (MFA) boundary refinement. Evaluation on manually annotated data shows that incorporating an LLM-based reasoning step improves turn-level speaker-role attribution from 62.70{\%} (acoustic diarization alone) to 95.68{\%}. Unlike prior uses of the MD archive, which focused on controlled sentence materials, MD{\_}NLP makes the spontaneous narrative component accessible for large-scale analysis. The resulting resource supports research on regional and socially conditioned variation, discourse structure, and corpus phonetics in Australian English. The proposed architecture is directly transferable to other legacy dialect archives, providing a practical pathway for transforming interview-based recordings into temporally aligned, speaker-consistent corpora."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="coats-2026-md">
<titleInfo>
<title>MD_NLP: Reconstructing an Australian English Heritage Dialect Corpus from the Mitchell-Delbridge Recordings through LLM-Assisted Speaker Attribution</title>
</titleInfo>
<name type="personal">
<namePart type="given">Steven</namePart>
<namePart type="family">Coats</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective</title>
</titleInfo>
<name type="personal">
<namePart type="given">Antonis</namePart>
<namePart type="family">Anastasopoulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stella</namePart>
<namePart type="family">Markantonatou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Angela</namePart>
<namePart type="family">Ralli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marcos</namePart>
<namePart type="family">Zampieri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stavros</namePart>
<namePart type="family">Bompolas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vivian</namePart>
<namePart type="family">Stamou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present MD_NLP, a discourse-annotated and georeferenced corpus derived from the Mitchell–Delbridge (MD) recordings, a foundational archive of mid-20th-century Australian English. The corpus comprises word-aligned narrative recordings from 7,735 secondary school pupils across 327 locations, enriched with structured sociodemographic metadata (e.g., sex, birthplace, parental background) and geocoded institutional coordinates. The narratives were reconstructed from archival audio using an integrated pipeline combining WhisperX-based automatic speech recognition, neural speaker diarization, LLM-assisted discourse-role correction, and Montreal Forced Aligner (MFA) boundary refinement. Evaluation on manually annotated data shows that incorporating an LLM-based reasoning step improves turn-level speaker-role attribution from 62.70% (acoustic diarization alone) to 95.68%. Unlike prior uses of the MD archive, which focused on controlled sentence materials, MD_NLP makes the spontaneous narrative component accessible for large-scale analysis. The resulting resource supports research on regional and socially conditioned variation, discourse structure, and corpus phonetics in Australian English. The proposed architecture is directly transferable to other legacy dialect archives, providing a practical pathway for transforming interview-based recordings into temporally aligned, speaker-consistent corpora.</abstract>
<identifier type="citekey">coats-2026-md</identifier>
<identifier type="doi">10.63317/3iga3zzsh92p</identifier>
<location>
<url>https://aclanthology.org/2026.dialres-1.3/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>24</start>
<end>32</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T MD_NLP: Reconstructing an Australian English Heritage Dialect Corpus from the Mitchell-Delbridge Recordings through LLM-Assisted Speaker Attribution
%A Coats, Steven
%Y Anastasopoulos, Antonis
%Y Markantonatou, Stella
%Y Ralli, Angela
%Y Zampieri, Marcos
%Y Bompolas, Stavros
%Y Stamou, Vivian
%S Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca
%F coats-2026-md
%X We present MD_NLP, a discourse-annotated and georeferenced corpus derived from the Mitchell–Delbridge (MD) recordings, a foundational archive of mid-20th-century Australian English. The corpus comprises word-aligned narrative recordings from 7,735 secondary school pupils across 327 locations, enriched with structured sociodemographic metadata (e.g., sex, birthplace, parental background) and geocoded institutional coordinates. The narratives were reconstructed from archival audio using an integrated pipeline combining WhisperX-based automatic speech recognition, neural speaker diarization, LLM-assisted discourse-role correction, and Montreal Forced Aligner (MFA) boundary refinement. Evaluation on manually annotated data shows that incorporating an LLM-based reasoning step improves turn-level speaker-role attribution from 62.70% (acoustic diarization alone) to 95.68%. Unlike prior uses of the MD archive, which focused on controlled sentence materials, MD_NLP makes the spontaneous narrative component accessible for large-scale analysis. The resulting resource supports research on regional and socially conditioned variation, discourse structure, and corpus phonetics in Australian English. The proposed architecture is directly transferable to other legacy dialect archives, providing a practical pathway for transforming interview-based recordings into temporally aligned, speaker-consistent corpora.
%R 10.63317/3iga3zzsh92p
%U https://aclanthology.org/2026.dialres-1.3/
%U https://doi.org/10.63317/3iga3zzsh92p
%P 24-32
Markdown (Informal)
[MD_NLP: Reconstructing an Australian English Heritage Dialect Corpus from the Mitchell-Delbridge Recordings through LLM-Assisted Speaker Attribution](https://aclanthology.org/2026.dialres-1.3/) (Coats, DialRes 2026)
ACL