@inproceedings{peng-2026-g,
title = "{G}{\&}{P}2{P}: A Multi-Source Approach to Grapheme-to-Phoneme Conversion",
author = "Peng, Chun-Yi Jerry",
editor = "Gorman, Kyle",
booktitle = "Proceedings of the Third Workshop on Computation and Written Language ({CAWL} 2026) @ {LREC} 2026",
month = jun,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cawl-1.9/",
doi = "10.63317/4f2x3fda6jj7",
pages = "89--94",
abstract = "Grapheme-to-phoneme (G2P) conversion plays a central role in speech technologies. This paper introduces G{\&}P2P, a multi-source framework that integrates multiple pronunciation dictionaries to enhance G2P modeling. We evaluate both expert-curated and crowd-sourced resources using attentive LSTM, pointer-generator LSTM, and transformer architectures. Results indicate that combining high-quality expert dictionaries yields substantial improvements, achieving an 11.26-point absolute (22{\%} relative) reduction in word error rate. In contrast, incorporating noisy crowd-sourced resources may degrade performance. Statistical analyses further suggest that dataset quality exerts a greater influence on outcomes than the choice of fusion strategy, offering practical guidance for the design of multi-source G2P systems."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="peng-2026-g">
<titleInfo>
<title>G&P2P: A Multi-Source Approach to Grapheme-to-Phoneme Conversion</title>
</titleInfo>
<name type="personal">
<namePart type="given">Chun-Yi</namePart>
<namePart type="given">Jerry</namePart>
<namePart type="family">Peng</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Third Workshop on Computation and Written Language (CAWL 2026) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Kyle</namePart>
<namePart type="family">Gorman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Grapheme-to-phoneme (G2P) conversion plays a central role in speech technologies. This paper introduces G&P2P, a multi-source framework that integrates multiple pronunciation dictionaries to enhance G2P modeling. We evaluate both expert-curated and crowd-sourced resources using attentive LSTM, pointer-generator LSTM, and transformer architectures. Results indicate that combining high-quality expert dictionaries yields substantial improvements, achieving an 11.26-point absolute (22% relative) reduction in word error rate. In contrast, incorporating noisy crowd-sourced resources may degrade performance. Statistical analyses further suggest that dataset quality exerts a greater influence on outcomes than the choice of fusion strategy, offering practical guidance for the design of multi-source G2P systems.</abstract>
<identifier type="citekey">peng-2026-g</identifier>
<identifier type="doi">10.63317/4f2x3fda6jj7</identifier>
<location>
<url>https://aclanthology.org/2026.cawl-1.9/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>89</start>
<end>94</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T G&P2P: A Multi-Source Approach to Grapheme-to-Phoneme Conversion
%A Peng, Chun-Yi Jerry
%Y Gorman, Kyle
%S Proceedings of the Third Workshop on Computation and Written Language (CAWL 2026) @ LREC 2026
%D 2026
%8 June
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F peng-2026-g
%X Grapheme-to-phoneme (G2P) conversion plays a central role in speech technologies. This paper introduces G&P2P, a multi-source framework that integrates multiple pronunciation dictionaries to enhance G2P modeling. We evaluate both expert-curated and crowd-sourced resources using attentive LSTM, pointer-generator LSTM, and transformer architectures. Results indicate that combining high-quality expert dictionaries yields substantial improvements, achieving an 11.26-point absolute (22% relative) reduction in word error rate. In contrast, incorporating noisy crowd-sourced resources may degrade performance. Statistical analyses further suggest that dataset quality exerts a greater influence on outcomes than the choice of fusion strategy, offering practical guidance for the design of multi-source G2P systems.
%R 10.63317/4f2x3fda6jj7
%U https://aclanthology.org/2026.cawl-1.9/
%U https://doi.org/10.63317/4f2x3fda6jj7
%P 89-94
Markdown (Informal)
[G&P2P: A Multi-Source Approach to Grapheme-to-Phoneme Conversion](https://aclanthology.org/2026.cawl-1.9/) (Peng, CAWL 2026)
ACL