@inproceedings{fischer-lameli-2026-german,
title = "{G}erman Dialects Across Situations, Generations, and Regions: The {REDE} corpus as an Oral Resource for {NLP}",
author = "Fischer, Hanna and
Lameli, Alfred",
editor = "Anastasopoulos, Antonis and
Markantonatou, Stella and
Ralli, Angela and
Zampieri, Marcos and
Bompolas, Stavros and
Stamou, Vivian",
booktitle = "Proceedings of the First Workshop on Dialects in {NLP} {---} A Resource Perspective",
month = may,
year = "2026",
address = "Palma de Mallorca",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.dialres-1.15/",
doi = "10.63317/4fe4dkefqah9",
pages = "144--152",
abstract = "Recent advances in speech and language technologies increasingly rely on large and diverse corpora that represent linguistic variation across dialect regions, communicative situations, and social speaker characteristics. While substantial resources are available for Standard German, comparable spoken corpora for German dialects have so far been largely lacking, limiting the development and evaluation of dialect-sensitive NLP systems. The REDE corpus addresses this gap by providing a methodologically uniform collection of spoken German for 148 locations that systematically covers all major dialect areas in Germany. It comprises contemporary recordings collected in multiple elicitation and interaction settings, capturing variation across speaking styles, situational contexts, and speaker generations. With more than 1,500 hours of speech and rich metadata on regional and social dimensions, the REDE corpus constitutes a large-scale oral resource suitable for both linguistic research and NLP applications. This paper presents the design, structure, and methodological foundations of the corpus and discusses its relevance for current speech technology requirements."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="fischer-lameli-2026-german">
<titleInfo>
<title>German Dialects Across Situations, Generations, and Regions: The REDE corpus as an Oral Resource for NLP</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hanna</namePart>
<namePart type="family">Fischer</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alfred</namePart>
<namePart type="family">Lameli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective</title>
</titleInfo>
<name type="personal">
<namePart type="given">Antonis</namePart>
<namePart type="family">Anastasopoulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stella</namePart>
<namePart type="family">Markantonatou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Angela</namePart>
<namePart type="family">Ralli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marcos</namePart>
<namePart type="family">Zampieri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stavros</namePart>
<namePart type="family">Bompolas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vivian</namePart>
<namePart type="family">Stamou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Recent advances in speech and language technologies increasingly rely on large and diverse corpora that represent linguistic variation across dialect regions, communicative situations, and social speaker characteristics. While substantial resources are available for Standard German, comparable spoken corpora for German dialects have so far been largely lacking, limiting the development and evaluation of dialect-sensitive NLP systems. The REDE corpus addresses this gap by providing a methodologically uniform collection of spoken German for 148 locations that systematically covers all major dialect areas in Germany. It comprises contemporary recordings collected in multiple elicitation and interaction settings, capturing variation across speaking styles, situational contexts, and speaker generations. With more than 1,500 hours of speech and rich metadata on regional and social dimensions, the REDE corpus constitutes a large-scale oral resource suitable for both linguistic research and NLP applications. This paper presents the design, structure, and methodological foundations of the corpus and discusses its relevance for current speech technology requirements.</abstract>
<identifier type="citekey">fischer-lameli-2026-german</identifier>
<identifier type="doi">10.63317/4fe4dkefqah9</identifier>
<location>
<url>https://aclanthology.org/2026.dialres-1.15/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>144</start>
<end>152</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T German Dialects Across Situations, Generations, and Regions: The REDE corpus as an Oral Resource for NLP
%A Fischer, Hanna
%A Lameli, Alfred
%Y Anastasopoulos, Antonis
%Y Markantonatou, Stella
%Y Ralli, Angela
%Y Zampieri, Marcos
%Y Bompolas, Stavros
%Y Stamou, Vivian
%S Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca
%F fischer-lameli-2026-german
%X Recent advances in speech and language technologies increasingly rely on large and diverse corpora that represent linguistic variation across dialect regions, communicative situations, and social speaker characteristics. While substantial resources are available for Standard German, comparable spoken corpora for German dialects have so far been largely lacking, limiting the development and evaluation of dialect-sensitive NLP systems. The REDE corpus addresses this gap by providing a methodologically uniform collection of spoken German for 148 locations that systematically covers all major dialect areas in Germany. It comprises contemporary recordings collected in multiple elicitation and interaction settings, capturing variation across speaking styles, situational contexts, and speaker generations. With more than 1,500 hours of speech and rich metadata on regional and social dimensions, the REDE corpus constitutes a large-scale oral resource suitable for both linguistic research and NLP applications. This paper presents the design, structure, and methodological foundations of the corpus and discusses its relevance for current speech technology requirements.
%R 10.63317/4fe4dkefqah9
%U https://aclanthology.org/2026.dialres-1.15/
%U https://doi.org/10.63317/4fe4dkefqah9
%P 144-152
Markdown (Informal)
[German Dialects Across Situations, Generations, and Regions: The REDE corpus as an Oral Resource for NLP](https://aclanthology.org/2026.dialres-1.15/) (Fischer & Lameli, DialRes 2026)
ACL