@inproceedings{bleaman-2026-automatic,
title = "Automatic Transcription of Holocaust Testimonies in {Y}iddish: Orthographic Comparison and Cross-Domain Validation",
author = "Bleaman, Isaac L.",
editor = "Anuradha, Isuri and
Wynne, Martin",
booktitle = "Proceedings of The Second Workshop on Holocaust Testimonies as Language Resources ({HTR}es)",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.htres-2.3/",
doi = "10.63317/24nawizn9q4x",
pages = "20--28",
abstract = "The digitization and computational processing of Holocaust testimony interviews are essential for the long-term preservation and accessibility of survivors' narratives. However, automatic speech recognition (ASR) for Yiddish{---}the primary language of most Holocaust victims and survivors{---}remains underdeveloped. This paper introduces the first ASR system for European Yiddish, focused on the Northeastern ({``}Lithuanian'') dialect and trained on Holocaust survivor testimonies from the Corpus of Spoken Yiddish in Europe (42 hours of speech segments from 60 survivors). A systematic comparison of CTC-based ASR models using transcripts with different orthographic representations reveals that a Hebrew-based phonemic system with precomposed Unicode is optimal, achieving a mean WER of 37.96{\%} compared to 59.40{\%} WER for romanized Yiddish and 99.67{\%} WER (catastrophic failure) for standard Yiddish spelled with decomposed Unicode. Cross-domain testing on Yiddish audiobooks provides additional support for a phonemic representation (27.07{\%} WER, 6.56{\%} CER). Together, the results suggest that automatic transcription developed from oral Holocaust testimonies can support further technological innovation in service of Yiddish-speaking communities."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="bleaman-2026-automatic">
<titleInfo>
<title>Automatic Transcription of Holocaust Testimonies in Yiddish: Orthographic Comparison and Cross-Domain Validation</title>
</titleInfo>
<name type="personal">
<namePart type="given">Isaac</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Bleaman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of The Second Workshop on Holocaust Testimonies as Language Resources (HTRes)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Isuri</namePart>
<namePart type="family">Anuradha</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Martin</namePart>
<namePart type="family">Wynne</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The digitization and computational processing of Holocaust testimony interviews are essential for the long-term preservation and accessibility of survivors’ narratives. However, automatic speech recognition (ASR) for Yiddish—the primary language of most Holocaust victims and survivors—remains underdeveloped. This paper introduces the first ASR system for European Yiddish, focused on the Northeastern (“Lithuanian”) dialect and trained on Holocaust survivor testimonies from the Corpus of Spoken Yiddish in Europe (42 hours of speech segments from 60 survivors). A systematic comparison of CTC-based ASR models using transcripts with different orthographic representations reveals that a Hebrew-based phonemic system with precomposed Unicode is optimal, achieving a mean WER of 37.96% compared to 59.40% WER for romanized Yiddish and 99.67% WER (catastrophic failure) for standard Yiddish spelled with decomposed Unicode. Cross-domain testing on Yiddish audiobooks provides additional support for a phonemic representation (27.07% WER, 6.56% CER). Together, the results suggest that automatic transcription developed from oral Holocaust testimonies can support further technological innovation in service of Yiddish-speaking communities.</abstract>
<identifier type="citekey">bleaman-2026-automatic</identifier>
<identifier type="doi">10.63317/24nawizn9q4x</identifier>
<location>
<url>https://aclanthology.org/2026.htres-2.3/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>20</start>
<end>28</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Automatic Transcription of Holocaust Testimonies in Yiddish: Orthographic Comparison and Cross-Domain Validation
%A Bleaman, Isaac L.
%Y Anuradha, Isuri
%Y Wynne, Martin
%S Proceedings of The Second Workshop on Holocaust Testimonies as Language Resources (HTRes)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F bleaman-2026-automatic
%X The digitization and computational processing of Holocaust testimony interviews are essential for the long-term preservation and accessibility of survivors’ narratives. However, automatic speech recognition (ASR) for Yiddish—the primary language of most Holocaust victims and survivors—remains underdeveloped. This paper introduces the first ASR system for European Yiddish, focused on the Northeastern (“Lithuanian”) dialect and trained on Holocaust survivor testimonies from the Corpus of Spoken Yiddish in Europe (42 hours of speech segments from 60 survivors). A systematic comparison of CTC-based ASR models using transcripts with different orthographic representations reveals that a Hebrew-based phonemic system with precomposed Unicode is optimal, achieving a mean WER of 37.96% compared to 59.40% WER for romanized Yiddish and 99.67% WER (catastrophic failure) for standard Yiddish spelled with decomposed Unicode. Cross-domain testing on Yiddish audiobooks provides additional support for a phonemic representation (27.07% WER, 6.56% CER). Together, the results suggest that automatic transcription developed from oral Holocaust testimonies can support further technological innovation in service of Yiddish-speaking communities.
%R 10.63317/24nawizn9q4x
%U https://aclanthology.org/2026.htres-2.3/
%U https://doi.org/10.63317/24nawizn9q4x
%P 20-28
Markdown (Informal)
[Automatic Transcription of Holocaust Testimonies in Yiddish: Orthographic Comparison and Cross-Domain Validation](https://aclanthology.org/2026.htres-2.3/) (Bleaman, htres 2026)
ACL