@inproceedings{yang-etal-2026-tg,
title = "{TG}-{ASR}: Translation-Guided Learning with Parallel Gated Cross Attention for Low-Resource Automatic Speech Recognition",
author = "Yang, ChengYeh and
Wang, Chien-Chun and
Chen, Li-Wei and
Lee, Hung-Shin and
Wang, Hsin-Min and
Chen, Berlin",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.790/",
doi = "10.63317/2ne34rzfnfgz",
pages = "10071--10081",
abstract = "Low-resource automatic speech recognition remains a critical challenge due to the scarcity of transcribed data for many languages.Taiwanese Hokkien exemplifies this problem as, although extensive speech content exists in television dramas and online videos, transcriptions are scarce and most available subtitles are in Mandarin.To address this gap, this paper presents TG-ASR for Taiwanese drama speech recognition, a translation-guided ASR framework that leverages multilingual translation embeddings to enhance recognition in low-resource conditions.The framework centers on the parallel gated cross-attention (PGCA) mechanism, which adaptively integrates embeddings from multiple auxiliary languages into the ASR decoder.This mechanism enables robust cross-linguistic semantic guidance while maintaining stable optimization and avoiding interference between languages.To support future research, we release YT-THDC, a 30-hour corpus of Taiwanese drama speech with aligned Mandarin subtitles and manually verified Taiwanese transcriptions.Extensive experiments and analysis identify which auxiliary languages most effectively improve Taiwanese ASR, achieving a 13.51{\%} relative reduction in character error rate and demonstrating the potential of translation-guided learning for underrepresented languages in real-world scenarios."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="yang-etal-2026-tg">
<titleInfo>
<title>TG-ASR: Translation-Guided Learning with Parallel Gated Cross Attention for Low-Resource Automatic Speech Recognition</title>
</titleInfo>
<name type="personal">
<namePart type="given">ChengYeh</namePart>
<namePart type="family">Yang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Chien-Chun</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Li-Wei</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hung-Shin</namePart>
<namePart type="family">Lee</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hsin-Min</namePart>
<namePart type="family">Wang</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Berlin</namePart>
<namePart type="family">Chen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Low-resource automatic speech recognition remains a critical challenge due to the scarcity of transcribed data for many languages.Taiwanese Hokkien exemplifies this problem as, although extensive speech content exists in television dramas and online videos, transcriptions are scarce and most available subtitles are in Mandarin.To address this gap, this paper presents TG-ASR for Taiwanese drama speech recognition, a translation-guided ASR framework that leverages multilingual translation embeddings to enhance recognition in low-resource conditions.The framework centers on the parallel gated cross-attention (PGCA) mechanism, which adaptively integrates embeddings from multiple auxiliary languages into the ASR decoder.This mechanism enables robust cross-linguistic semantic guidance while maintaining stable optimization and avoiding interference between languages.To support future research, we release YT-THDC, a 30-hour corpus of Taiwanese drama speech with aligned Mandarin subtitles and manually verified Taiwanese transcriptions.Extensive experiments and analysis identify which auxiliary languages most effectively improve Taiwanese ASR, achieving a 13.51% relative reduction in character error rate and demonstrating the potential of translation-guided learning for underrepresented languages in real-world scenarios.</abstract>
<identifier type="citekey">yang-etal-2026-tg</identifier>
<identifier type="doi">10.63317/2ne34rzfnfgz</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.790/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>10071</start>
<end>10081</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T TG-ASR: Translation-Guided Learning with Parallel Gated Cross Attention for Low-Resource Automatic Speech Recognition
%A Yang, ChengYeh
%A Wang, Chien-Chun
%A Chen, Li-Wei
%A Lee, Hung-Shin
%A Wang, Hsin-Min
%A Chen, Berlin
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F yang-etal-2026-tg
%X Low-resource automatic speech recognition remains a critical challenge due to the scarcity of transcribed data for many languages.Taiwanese Hokkien exemplifies this problem as, although extensive speech content exists in television dramas and online videos, transcriptions are scarce and most available subtitles are in Mandarin.To address this gap, this paper presents TG-ASR for Taiwanese drama speech recognition, a translation-guided ASR framework that leverages multilingual translation embeddings to enhance recognition in low-resource conditions.The framework centers on the parallel gated cross-attention (PGCA) mechanism, which adaptively integrates embeddings from multiple auxiliary languages into the ASR decoder.This mechanism enables robust cross-linguistic semantic guidance while maintaining stable optimization and avoiding interference between languages.To support future research, we release YT-THDC, a 30-hour corpus of Taiwanese drama speech with aligned Mandarin subtitles and manually verified Taiwanese transcriptions.Extensive experiments and analysis identify which auxiliary languages most effectively improve Taiwanese ASR, achieving a 13.51% relative reduction in character error rate and demonstrating the potential of translation-guided learning for underrepresented languages in real-world scenarios.
%R 10.63317/2ne34rzfnfgz
%U https://aclanthology.org/2026.lrec-1.790/
%U https://doi.org/10.63317/2ne34rzfnfgz
%P 10071-10081
Markdown (Informal)
[TG-ASR: Translation-Guided Learning with Parallel Gated Cross Attention for Low-Resource Automatic Speech Recognition](https://aclanthology.org/2026.lrec-1.790/) (Yang et al., LREC 2026)
ACL