@inproceedings{konstantinidou-etal-2026-speech,
title = "A Speech Resource for the {P}ontic {G}reek Dialect: Transcription Choices and Baseline {ASR} Evaluation",
author = "Konstantinidou, Rodanna and
Tsoukala, Chara and
Stamou, Vivian and
Giouli, Voula and
Markantonatou, Stella",
editor = "Anastasopoulos, Antonis and
Markantonatou, Stella and
Ralli, Angela and
Zampieri, Marcos and
Bompolas, Stavros and
Stamou, Vivian",
booktitle = "Proceedings of the First Workshop on Dialects in {NLP} {---} A Resource Perspective",
month = may,
year = "2026",
address = "Palma de Mallorca",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.dialres-1.29/",
doi = "10.63317/3rq2murx2626",
pages = "300--307",
abstract = "Pontic Greek is a living but endangered Modern Greek dialect that lacks publicly available AI-oriented speech resources and ASR benchmarks. This work reports on the first systematic inference-only (zero-shot) ASR evaluation on authentic Pontic speech. Progress on Pontic ASR is hindered by two coupled challenges: the scarcity of transcribed speech data and the absence of a standardized orthography, which makes it difficult to create consistent reference transcriptions for evaluation. We address these challenges by releasing a new speech corpus of contemporary Pontic as spoken in Northern Greece, derived from natural conversations and provided with manual, utterance-level, time-aligned transcriptions. To reduce annotator bias and increase practical usability, we collect community evidence on written-form preferences via a small questionnaire and use the observed patterns to guide a consistent Greek-script transcription scheme. We use this corpus to perform inference-only (zero-shot) ASR evaluation, benchmarking four state-of-the-art pretrained speech recognition models under a unified evaluation protocol. Results show that zero-shot recognition remains challenging, establishing baseline figures and underscoring the need for dialect-specific data and adaptation."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="konstantinidou-etal-2026-speech">
<titleInfo>
<title>A Speech Resource for the Pontic Greek Dialect: Transcription Choices and Baseline ASR Evaluation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Rodanna</namePart>
<namePart type="family">Konstantinidou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chara</namePart>
<namePart type="family">Tsoukala</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vivian</namePart>
<namePart type="family">Stamou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Voula</namePart>
<namePart type="family">Giouli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stella</namePart>
<namePart type="family">Markantonatou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective</title>
</titleInfo>
<name type="personal">
<namePart type="given">Antonis</namePart>
<namePart type="family">Anastasopoulos</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stella</namePart>
<namePart type="family">Markantonatou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Angela</namePart>
<namePart type="family">Ralli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marcos</namePart>
<namePart type="family">Zampieri</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stavros</namePart>
<namePart type="family">Bompolas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vivian</namePart>
<namePart type="family">Stamou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Pontic Greek is a living but endangered Modern Greek dialect that lacks publicly available AI-oriented speech resources and ASR benchmarks. This work reports on the first systematic inference-only (zero-shot) ASR evaluation on authentic Pontic speech. Progress on Pontic ASR is hindered by two coupled challenges: the scarcity of transcribed speech data and the absence of a standardized orthography, which makes it difficult to create consistent reference transcriptions for evaluation. We address these challenges by releasing a new speech corpus of contemporary Pontic as spoken in Northern Greece, derived from natural conversations and provided with manual, utterance-level, time-aligned transcriptions. To reduce annotator bias and increase practical usability, we collect community evidence on written-form preferences via a small questionnaire and use the observed patterns to guide a consistent Greek-script transcription scheme. We use this corpus to perform inference-only (zero-shot) ASR evaluation, benchmarking four state-of-the-art pretrained speech recognition models under a unified evaluation protocol. Results show that zero-shot recognition remains challenging, establishing baseline figures and underscoring the need for dialect-specific data and adaptation.</abstract>
<identifier type="citekey">konstantinidou-etal-2026-speech</identifier>
<identifier type="doi">10.63317/3rq2murx2626</identifier>
<location>
<url>https://aclanthology.org/2026.dialres-1.29/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>300</start>
<end>307</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T A Speech Resource for the Pontic Greek Dialect: Transcription Choices and Baseline ASR Evaluation
%A Konstantinidou, Rodanna
%A Tsoukala, Chara
%A Stamou, Vivian
%A Giouli, Voula
%A Markantonatou, Stella
%Y Anastasopoulos, Antonis
%Y Markantonatou, Stella
%Y Ralli, Angela
%Y Zampieri, Marcos
%Y Bompolas, Stavros
%Y Stamou, Vivian
%S Proceedings of the First Workshop on Dialects in NLP — A Resource Perspective
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca
%F konstantinidou-etal-2026-speech
%X Pontic Greek is a living but endangered Modern Greek dialect that lacks publicly available AI-oriented speech resources and ASR benchmarks. This work reports on the first systematic inference-only (zero-shot) ASR evaluation on authentic Pontic speech. Progress on Pontic ASR is hindered by two coupled challenges: the scarcity of transcribed speech data and the absence of a standardized orthography, which makes it difficult to create consistent reference transcriptions for evaluation. We address these challenges by releasing a new speech corpus of contemporary Pontic as spoken in Northern Greece, derived from natural conversations and provided with manual, utterance-level, time-aligned transcriptions. To reduce annotator bias and increase practical usability, we collect community evidence on written-form preferences via a small questionnaire and use the observed patterns to guide a consistent Greek-script transcription scheme. We use this corpus to perform inference-only (zero-shot) ASR evaluation, benchmarking four state-of-the-art pretrained speech recognition models under a unified evaluation protocol. Results show that zero-shot recognition remains challenging, establishing baseline figures and underscoring the need for dialect-specific data and adaptation.
%R 10.63317/3rq2murx2626
%U https://aclanthology.org/2026.dialres-1.29/
%U https://doi.org/10.63317/3rq2murx2626
%P 300-307
Markdown (Informal)
[A Speech Resource for the Pontic Greek Dialect: Transcription Choices and Baseline ASR Evaluation](https://aclanthology.org/2026.dialres-1.29/) (Konstantinidou et al., DialRes 2026)
ACL