@inproceedings{ligeti-nagy-szabo-2026-towards,
title = "Towards an interoperable {H}ungarian historical newspaper corpus",
author = "Ligeti-Nagy, No{\'e}mi and
Szab{\'o}, Henrietta",
editor = "Ogrodniczuk, Maciej and
Osenova, Petya and
Wissik, Tanja",
booktitle = "Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.pressmint-1.9/",
doi = "10.63317/33jwhn35vqjk",
pages = "50--55",
abstract = "PressMint is a CLARIN initiative that aims to build multilingual, comparable and interoperable corpora of historical newspapers. For Hungarian, the main challenge is not a lack of material but fragmentation: newspapers are distributed across several portals, with heterogeneous metadata, access paths and OCR quality. This extended abstract reports the current status of the Hungarian PressMint subcorpus, focusing on the 19th century and the early 20th century (roughly 1800{--}1920). We describe two project artefacts already used in practice: a structured source inventory and a validation-driven repository. We summarise source scouting across Europeana, Hungaricana, OSZK{--}EPA, DiFMOE and related portals, including a curated 12-title Hungaricana manual-download pilot list with explicit target coverage periods. We then outline a reproducible pipeline for acquisition, OCR, layout analysis and conversion to PressMint-compatible TEI with facsimile linkage. Finally, we specify near-term deliverables for a first Hungarian release candidate and the evaluation steps planned for OCR and layout processing."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="ligeti-nagy-szabo-2026-towards">
<titleInfo>
<title>Towards an interoperable Hungarian historical newspaper corpus</title>
</titleInfo>
<name type="personal">
<namePart type="given">Noémi</namePart>
<namePart type="family">Ligeti-Nagy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henrietta</namePart>
<namePart type="family">Szabó</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Maciej</namePart>
<namePart type="family">Ogrodniczuk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Petya</namePart>
<namePart type="family">Osenova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tanja</namePart>
<namePart type="family">Wissik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>PressMint is a CLARIN initiative that aims to build multilingual, comparable and interoperable corpora of historical newspapers. For Hungarian, the main challenge is not a lack of material but fragmentation: newspapers are distributed across several portals, with heterogeneous metadata, access paths and OCR quality. This extended abstract reports the current status of the Hungarian PressMint subcorpus, focusing on the 19th century and the early 20th century (roughly 1800–1920). We describe two project artefacts already used in practice: a structured source inventory and a validation-driven repository. We summarise source scouting across Europeana, Hungaricana, OSZK–EPA, DiFMOE and related portals, including a curated 12-title Hungaricana manual-download pilot list with explicit target coverage periods. We then outline a reproducible pipeline for acquisition, OCR, layout analysis and conversion to PressMint-compatible TEI with facsimile linkage. Finally, we specify near-term deliverables for a first Hungarian release candidate and the evaluation steps planned for OCR and layout processing.</abstract>
<identifier type="citekey">ligeti-nagy-szabo-2026-towards</identifier>
<identifier type="doi">10.63317/33jwhn35vqjk</identifier>
<location>
<url>https://aclanthology.org/2026.pressmint-1.9/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>50</start>
<end>55</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Towards an interoperable Hungarian historical newspaper corpus
%A Ligeti-Nagy, Noémi
%A Szabó, Henrietta
%Y Ogrodniczuk, Maciej
%Y Osenova, Petya
%Y Wissik, Tanja
%S Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca, Spain
%F ligeti-nagy-szabo-2026-towards
%X PressMint is a CLARIN initiative that aims to build multilingual, comparable and interoperable corpora of historical newspapers. For Hungarian, the main challenge is not a lack of material but fragmentation: newspapers are distributed across several portals, with heterogeneous metadata, access paths and OCR quality. This extended abstract reports the current status of the Hungarian PressMint subcorpus, focusing on the 19th century and the early 20th century (roughly 1800–1920). We describe two project artefacts already used in practice: a structured source inventory and a validation-driven repository. We summarise source scouting across Europeana, Hungaricana, OSZK–EPA, DiFMOE and related portals, including a curated 12-title Hungaricana manual-download pilot list with explicit target coverage periods. We then outline a reproducible pipeline for acquisition, OCR, layout analysis and conversion to PressMint-compatible TEI with facsimile linkage. Finally, we specify near-term deliverables for a first Hungarian release candidate and the evaluation steps planned for OCR and layout processing.
%R 10.63317/33jwhn35vqjk
%U https://aclanthology.org/2026.pressmint-1.9/
%U https://doi.org/10.63317/33jwhn35vqjk
%P 50-55
Markdown (Informal)
[Towards an interoperable Hungarian historical newspaper corpus](https://aclanthology.org/2026.pressmint-1.9/) (Ligeti-Nagy & Szabó, PressMint 2026)
ACL