@inproceedings{blevins-2026-systematic,
title = "Systematic Normalization of Spoken Mixed-Language, Mixed-Dialect Data",
author = "Blevins, Margaret",
editor = "Anastasopoulos, Antonis and
Markantonatou, Stella and
Ralli, Angela and
Zampieri, Marcos and
Bompolas, Stavros and
Stamou, Vivian",
booktitle = "Proceedings of the First Workshop on Dialects in {NLP} {---} A Resource Perspective",
month = may,
year = "2026",
address = "Palma de Mallorca",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.dialres-1.6/",
doi = "10.63317/3bv9dmxr24p6",
pages = "58--69",
abstract = "Literary transcriptions of spoken language often deviate from standard, written language. These variations can lead to higher than desirable error rates in NLP processing. This is particularly the case for spoken data of low resource varieties, including dialects and contact varieties of higher resource languages. This paper outlines a proposal for the systematic dialect-to-standard normalization of spoken language from language contact and dialect contact situations. This system is then tested on the Texas German Sample Corpus ({\textasciitilde}13 hours), a set of audio and transcripts of Texas German conversations. Texas German is an umbrella term for a set of a heritage varieties of German spoken in Texas, USA that descend from multiple German dialects and that have been in contact with English for 150+ years. The proposed normalization system, along with the accompanying language-tagging system, can act as a starting point for other projects interested in normalizing their mixed variety data."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="blevins-2026-systematic">
<titleInfo>
<title>Systematic Normalization of Spoken Mixed-Language, Mixed-Dialect Data</title>
</titleInfo>
<name type="personal">
<namePart type="given">Margaret</namePart>
<namePart type="family">Blevins</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective</title>
</titleInfo>
<name type="personal">
<namePart type="given">Antonis</namePart>
<namePart type="family">Anastasopoulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stella</namePart>
<namePart type="family">Markantonatou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Angela</namePart>
<namePart type="family">Ralli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marcos</namePart>
<namePart type="family">Zampieri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stavros</namePart>
<namePart type="family">Bompolas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vivian</namePart>
<namePart type="family">Stamou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Literary transcriptions of spoken language often deviate from standard, written language. These variations can lead to higher than desirable error rates in NLP processing. This is particularly the case for spoken data of low resource varieties, including dialects and contact varieties of higher resource languages. This paper outlines a proposal for the systematic dialect-to-standard normalization of spoken language from language contact and dialect contact situations. This system is then tested on the Texas German Sample Corpus (~13 hours), a set of audio and transcripts of Texas German conversations. Texas German is an umbrella term for a set of a heritage varieties of German spoken in Texas, USA that descend from multiple German dialects and that have been in contact with English for 150+ years. The proposed normalization system, along with the accompanying language-tagging system, can act as a starting point for other projects interested in normalizing their mixed variety data.</abstract>
<identifier type="citekey">blevins-2026-systematic</identifier>
<identifier type="doi">10.63317/3bv9dmxr24p6</identifier>
<location>
<url>https://aclanthology.org/2026.dialres-1.6/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>58</start>
<end>69</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Systematic Normalization of Spoken Mixed-Language, Mixed-Dialect Data
%A Blevins, Margaret
%Y Anastasopoulos, Antonis
%Y Markantonatou, Stella
%Y Ralli, Angela
%Y Zampieri, Marcos
%Y Bompolas, Stavros
%Y Stamou, Vivian
%S Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca
%F blevins-2026-systematic
%X Literary transcriptions of spoken language often deviate from standard, written language. These variations can lead to higher than desirable error rates in NLP processing. This is particularly the case for spoken data of low resource varieties, including dialects and contact varieties of higher resource languages. This paper outlines a proposal for the systematic dialect-to-standard normalization of spoken language from language contact and dialect contact situations. This system is then tested on the Texas German Sample Corpus (~13 hours), a set of audio and transcripts of Texas German conversations. Texas German is an umbrella term for a set of a heritage varieties of German spoken in Texas, USA that descend from multiple German dialects and that have been in contact with English for 150+ years. The proposed normalization system, along with the accompanying language-tagging system, can act as a starting point for other projects interested in normalizing their mixed variety data.
%R 10.63317/3bv9dmxr24p6
%U https://aclanthology.org/2026.dialres-1.6/
%U https://doi.org/10.63317/3bv9dmxr24p6
%P 58-69
Markdown (Informal)
[Systematic Normalization of Spoken Mixed-Language, Mixed-Dialect Data](https://aclanthology.org/2026.dialres-1.6/) (Blevins, DialRes 2026)
ACL