@inproceedings{franzini-ducceschi-2026-south,
title = "South Tyrolean Dialect-to-Standard Speech Translation: A Resource",
author = "Franzini, Greta H. and
Ducceschi, Luca",
editor = "Anastasopoulos, Antonis and
Markantonatou, Stella and
Ralli, Angela and
Zampieri, Marcos and
Bompolas, Stavros and
Stamou, Vivian",
booktitle = "Proceedings of the First Workshop on Dialects in {NLP} {---} A Resource Perspective",
month = may,
year = "2026",
address = "Palma de Mallorca",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.dialres-1.19/",
doi = "10.63317/3visgk9f8s7z",
pages = "188--194",
abstract = "This paper presents a developing oral resource for South Tyrolean, a German dialect spoken in Northern Italy. The dialect is ubiquitous in spoken communication but lacks a standardised orthography. In this context, strict transcription into dialect is of limited to no utility to the local community. Instead, there is a distinct and strong demand for technology capable of directly translating spoken dialect into Standard German. To address this specific need, we introduce a dynamic, incrementally growing dataset designed to fine-tune ASR models for this translation task. Our corpus aggregates diverse sources, including media and research interviews, totalling over 13 hours of aligned audio. We describe a collaborative workflow where community partners contribute audio archives in exchange for automated transcriptions, creating a virtuous cycle of data improvement. Additionally, we detail our iterative model fine-tuning strategy, data collection challenges and the resulting improvements in model performance."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="franzini-ducceschi-2026-south">
<titleInfo>
<title>South Tyrolean Dialect-to-Standard Speech Translation: A Resource</title>
</titleInfo>
<name type="personal">
<namePart type="given">Greta</namePart>
<namePart type="given">H</namePart>
<namePart type="family">Franzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Luca</namePart>
<namePart type="family">Ducceschi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective</title>
</titleInfo>
<name type="personal">
<namePart type="given">Antonis</namePart>
<namePart type="family">Anastasopoulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stella</namePart>
<namePart type="family">Markantonatou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Angela</namePart>
<namePart type="family">Ralli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marcos</namePart>
<namePart type="family">Zampieri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stavros</namePart>
<namePart type="family">Bompolas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vivian</namePart>
<namePart type="family">Stamou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper presents a developing oral resource for South Tyrolean, a German dialect spoken in Northern Italy. The dialect is ubiquitous in spoken communication but lacks a standardised orthography. In this context, strict transcription into dialect is of limited to no utility to the local community. Instead, there is a distinct and strong demand for technology capable of directly translating spoken dialect into Standard German. To address this specific need, we introduce a dynamic, incrementally growing dataset designed to fine-tune ASR models for this translation task. Our corpus aggregates diverse sources, including media and research interviews, totalling over 13 hours of aligned audio. We describe a collaborative workflow where community partners contribute audio archives in exchange for automated transcriptions, creating a virtuous cycle of data improvement. Additionally, we detail our iterative model fine-tuning strategy, data collection challenges and the resulting improvements in model performance.</abstract>
<identifier type="citekey">franzini-ducceschi-2026-south</identifier>
<identifier type="doi">10.63317/3visgk9f8s7z</identifier>
<location>
<url>https://aclanthology.org/2026.dialres-1.19/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>188</start>
<end>194</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T South Tyrolean Dialect-to-Standard Speech Translation: A Resource
%A Franzini, Greta H.
%A Ducceschi, Luca
%Y Anastasopoulos, Antonis
%Y Markantonatou, Stella
%Y Ralli, Angela
%Y Zampieri, Marcos
%Y Bompolas, Stavros
%Y Stamou, Vivian
%S Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca
%F franzini-ducceschi-2026-south
%X This paper presents a developing oral resource for South Tyrolean, a German dialect spoken in Northern Italy. The dialect is ubiquitous in spoken communication but lacks a standardised orthography. In this context, strict transcription into dialect is of limited to no utility to the local community. Instead, there is a distinct and strong demand for technology capable of directly translating spoken dialect into Standard German. To address this specific need, we introduce a dynamic, incrementally growing dataset designed to fine-tune ASR models for this translation task. Our corpus aggregates diverse sources, including media and research interviews, totalling over 13 hours of aligned audio. We describe a collaborative workflow where community partners contribute audio archives in exchange for automated transcriptions, creating a virtuous cycle of data improvement. Additionally, we detail our iterative model fine-tuning strategy, data collection challenges and the resulting improvements in model performance.
%R 10.63317/3visgk9f8s7z
%U https://aclanthology.org/2026.dialres-1.19/
%U https://doi.org/10.63317/3visgk9f8s7z
%P 188-194
Markdown (Informal)
[South Tyrolean Dialect-to-Standard Speech Translation: A Resource](https://aclanthology.org/2026.dialres-1.19/) (Franzini & Ducceschi, DialRes 2026)
ACL