@inproceedings{bruckner-etal-2026-oral,
title = "From Oral History to Structured Data: The {M}alach{NER} Dataset",
author = {Br{\"u}ckner, Christopher and
Roginer Hofmeister, Karin and
Koci{\'a}n, Ji{\v{r}}{\'i} and
Pecina, Pavel},
editor = "Anuradha, Isuri and
Wynne, Martin",
booktitle = "Proceedings of The Second Workshop on Holocaust Testimonies as Language Resources ({HTR}es)",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.htres-2.7/",
doi = "10.63317/3zyb8y48bdnk",
pages = "59--65",
abstract = "We present MalachNER, a new multilingual dataset for Named Entity Recognition (NER) in testimonies of Holocaust survivors. MalachNER has been sourced from different archives and annotated based on comprehensive domain-specific guidelines refined by a collaboration of international experts. Covering 10 European languages, differs significantly from previously released datasets: It is primarily based on noisy, verbatim transcribed speech, rather than on digitized written documents. These transcripts are characterized, among other challenges, by fillers, dialectal speech, and in-line annotations indicating incomprehensible words, which are not commonly encountered in other datasets. However, large volumes of yet unprocessed oral history make such a dataset a necessity. In addition to the description of the dataset and its annotation guidelines, we show with baseline experiments that MalachNER is complementary with previously released data, and the key to training domain-specific language models that generalize well to written and oral testimony alike, achieving state-of-the-art performance on both types of documents."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="bruckner-etal-2026-oral">
<titleInfo>
<title>From Oral History to Structured Data: The MalachNER Dataset</title>
</titleInfo>
<name type="personal">
<namePart type="given">Christopher</namePart>
<namePart type="family">Brückner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Karin</namePart>
<namePart type="family">Roginer Hofmeister</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jiří</namePart>
<namePart type="family">Kocián</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pavel</namePart>
<namePart type="family">Pecina</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of The Second Workshop on Holocaust Testimonies as Language Resources (HTRes)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Isuri</namePart>
<namePart type="family">Anuradha</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Martin</namePart>
<namePart type="family">Wynne</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present MalachNER, a new multilingual dataset for Named Entity Recognition (NER) in testimonies of Holocaust survivors. MalachNER has been sourced from different archives and annotated based on comprehensive domain-specific guidelines refined by a collaboration of international experts. Covering 10 European languages, differs significantly from previously released datasets: It is primarily based on noisy, verbatim transcribed speech, rather than on digitized written documents. These transcripts are characterized, among other challenges, by fillers, dialectal speech, and in-line annotations indicating incomprehensible words, which are not commonly encountered in other datasets. However, large volumes of yet unprocessed oral history make such a dataset a necessity. In addition to the description of the dataset and its annotation guidelines, we show with baseline experiments that MalachNER is complementary with previously released data, and the key to training domain-specific language models that generalize well to written and oral testimony alike, achieving state-of-the-art performance on both types of documents.</abstract>
<identifier type="citekey">bruckner-etal-2026-oral</identifier>
<identifier type="doi">10.63317/3zyb8y48bdnk</identifier>
<location>
<url>https://aclanthology.org/2026.htres-2.7/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>59</start>
<end>65</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T From Oral History to Structured Data: The MalachNER Dataset
%A Brückner, Christopher
%A Roginer Hofmeister, Karin
%A Kocián, Jiří
%A Pecina, Pavel
%Y Anuradha, Isuri
%Y Wynne, Martin
%S Proceedings of The Second Workshop on Holocaust Testimonies as Language Resources (HTRes)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F bruckner-etal-2026-oral
%X We present MalachNER, a new multilingual dataset for Named Entity Recognition (NER) in testimonies of Holocaust survivors. MalachNER has been sourced from different archives and annotated based on comprehensive domain-specific guidelines refined by a collaboration of international experts. Covering 10 European languages, differs significantly from previously released datasets: It is primarily based on noisy, verbatim transcribed speech, rather than on digitized written documents. These transcripts are characterized, among other challenges, by fillers, dialectal speech, and in-line annotations indicating incomprehensible words, which are not commonly encountered in other datasets. However, large volumes of yet unprocessed oral history make such a dataset a necessity. In addition to the description of the dataset and its annotation guidelines, we show with baseline experiments that MalachNER is complementary with previously released data, and the key to training domain-specific language models that generalize well to written and oral testimony alike, achieving state-of-the-art performance on both types of documents.
%R 10.63317/3zyb8y48bdnk
%U https://aclanthology.org/2026.htres-2.7/
%U https://doi.org/10.63317/3zyb8y48bdnk
%P 59-65
Markdown (Informal)
[From Oral History to Structured Data: The MalachNER Dataset](https://aclanthology.org/2026.htres-2.7/) (Brückner et al., htres 2026)
ACL
- Christopher Brückner, Karin Roginer Hofmeister, Jiří Kocián, and Pavel Pecina. 2026. From Oral History to Structured Data: The MalachNER Dataset. In Proceedings of The Second Workshop on Holocaust Testimonies as Language Resources (HTRes), pages 59–65, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).