@inproceedings{senyk-etal-2025-context,
title = "Context-Aware Lexical Stress Prediction and Phonemization for {U}krainian {TTS} Systems",
author = "Senyk, Anastasiia and
Lukianchuk, Mykhailo and
Robeiko, Valentyna and
Paniv, Yurii",
editor = "Romanyshyn, Mariana",
booktitle = "Proceedings of the Fourth Ukrainian Natural Language Processing Workshop (UNLP 2025)",
month = jul,
year = "2025",
address = "Vienna, Austria (online)",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2025.unlp-1.11/",
doi = "10.18653/v1/2025.unlp-1.11",
pages = "96--104",
ISBN = "979-8-89176-269-5",
abstract = "Text preprocessing is a fundamental component of high-quality speech synthesis. This work presents a novel rule-based phonemizer combined with a sentence-level lexical stress prediction model to improve phonetic accuracy and prosody prediction in the text-to-speech pipelines. We also introduce a new benchmark dataset with annotated stress patterns designed for evaluating lexical stress prediction systems at the sentence level.Experimental results demonstrate that the proposed phonemizer achieves a 1.23{\%} word error rate on a manually constructed pronunciation dataset, while the lexical stress prediction pipeline shows results close to dictionary-based methods, outperforming existing neural network solutions."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="senyk-etal-2025-context">
<titleInfo>
<title>Context-Aware Lexical Stress Prediction and Phonemization for Ukrainian TTS Systems</title>
</titleInfo>
<name type="personal">
<namePart type="given">Anastasiia</namePart>
<namePart type="family">Senyk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mykhailo</namePart>
<namePart type="family">Lukianchuk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Valentyna</namePart>
<namePart type="family">Robeiko</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Yurii</namePart>
<namePart type="family">Paniv</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2025-07</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fourth Ukrainian Natural Language Processing Workshop (UNLP 2025)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mariana</namePart>
<namePart type="family">Romanyshyn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Vienna, Austria (online)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">979-8-89176-269-5</identifier>
</relatedItem>
<abstract>Text preprocessing is a fundamental component of high-quality speech synthesis. This work presents a novel rule-based phonemizer combined with a sentence-level lexical stress prediction model to improve phonetic accuracy and prosody prediction in the text-to-speech pipelines. We also introduce a new benchmark dataset with annotated stress patterns designed for evaluating lexical stress prediction systems at the sentence level.Experimental results demonstrate that the proposed phonemizer achieves a 1.23% word error rate on a manually constructed pronunciation dataset, while the lexical stress prediction pipeline shows results close to dictionary-based methods, outperforming existing neural network solutions.</abstract>
<identifier type="citekey">senyk-etal-2025-context</identifier>
<identifier type="doi">10.18653/v1/2025.unlp-1.11</identifier>
<location>
<url>https://aclanthology.org/2025.unlp-1.11/</url>
</location>
<part>
<date>2025-07</date>
<extent unit="page">
<start>96</start>
<end>104</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Context-Aware Lexical Stress Prediction and Phonemization for Ukrainian TTS Systems
%A Senyk, Anastasiia
%A Lukianchuk, Mykhailo
%A Robeiko, Valentyna
%A Paniv, Yurii
%Y Romanyshyn, Mariana
%S Proceedings of the Fourth Ukrainian Natural Language Processing Workshop (UNLP 2025)
%D 2025
%8 July
%I Association for Computational Linguistics
%C Vienna, Austria (online)
%@ 979-8-89176-269-5
%F senyk-etal-2025-context
%X Text preprocessing is a fundamental component of high-quality speech synthesis. This work presents a novel rule-based phonemizer combined with a sentence-level lexical stress prediction model to improve phonetic accuracy and prosody prediction in the text-to-speech pipelines. We also introduce a new benchmark dataset with annotated stress patterns designed for evaluating lexical stress prediction systems at the sentence level.Experimental results demonstrate that the proposed phonemizer achieves a 1.23% word error rate on a manually constructed pronunciation dataset, while the lexical stress prediction pipeline shows results close to dictionary-based methods, outperforming existing neural network solutions.
%R 10.18653/v1/2025.unlp-1.11
%U https://aclanthology.org/2025.unlp-1.11/
%U https://doi.org/10.18653/v1/2025.unlp-1.11
%P 96-104
Markdown (Informal)
[Context-Aware Lexical Stress Prediction and Phonemization for Ukrainian TTS Systems](https://aclanthology.org/2025.unlp-1.11/) (Senyk et al., UNLP 2025)
ACL