@inproceedings{erjavec-etal-2026-pressmint,
title = "{P}ress{M}int: Towards Interoperable Corpora of Historical Newspapers",
author = "Erjavec, Toma{\v{z}} and
Kopp, Maty{\'a}{\v{s}} and
Ogrodniczuk, Maciej and
Osenova, Petya and
Rigau, German",
editor = "Ogrodniczuk, Maciej and
Osenova, Petya and
Wissik, Tanja",
booktitle = "Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.pressmint-1.1/",
doi = "10.63317/55w7588oukim",
pages = "1--5",
abstract = "This paper presents Project X (name anonymized for review), an ongoing initiative to compile a multilingual, comparable, annotated, translated, and interoperable collection of European historical newspaper corpora. Spanning 17 countries and covering 15 languages, the project addresses a key shortcoming of existing newspaper resources: their lack of interoperability, which limits cross-lingual and transnational research. Building on the infrastructure and experience of the ParlaMint projects, the project adapts established encoding guidelines, validation workflows, and open-source tools to historical newspaper data. We outline the overall project architecture, the corpus encoding scheme, and the GitHub-based framework supporting collaborative development and quality control. The paper further describes the sample linguistic annotation pipeline, including OCR correction, text normalisation, and annotation within the Universal Dependencies framework, with attention to challenges posed by historical language varieties. The resulting FAIR, openly available corpora are intended to support comparative, diachronic research across the humanities and social sciences."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="erjavec-etal-2026-pressmint">
<titleInfo>
<title>PressMint: Towards Interoperable Corpora of Historical Newspapers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Tomaž</namePart>
<namePart type="family">Erjavec</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Matyáš</namePart>
<namePart type="family">Kopp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Maciej</namePart>
<namePart type="family">Ogrodniczuk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Petya</namePart>
<namePart type="family">Osenova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">German</namePart>
<namePart type="family">Rigau</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Maciej</namePart>
<namePart type="family">Ogrodniczuk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Petya</namePart>
<namePart type="family">Osenova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tanja</namePart>
<namePart type="family">Wissik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper presents Project X (name anonymized for review), an ongoing initiative to compile a multilingual, comparable, annotated, translated, and interoperable collection of European historical newspaper corpora. Spanning 17 countries and covering 15 languages, the project addresses a key shortcoming of existing newspaper resources: their lack of interoperability, which limits cross-lingual and transnational research. Building on the infrastructure and experience of the ParlaMint projects, the project adapts established encoding guidelines, validation workflows, and open-source tools to historical newspaper data. We outline the overall project architecture, the corpus encoding scheme, and the GitHub-based framework supporting collaborative development and quality control. The paper further describes the sample linguistic annotation pipeline, including OCR correction, text normalisation, and annotation within the Universal Dependencies framework, with attention to challenges posed by historical language varieties. The resulting FAIR, openly available corpora are intended to support comparative, diachronic research across the humanities and social sciences.</abstract>
<identifier type="citekey">erjavec-etal-2026-pressmint</identifier>
<identifier type="doi">10.63317/55w7588oukim</identifier>
<location>
<url>https://aclanthology.org/2026.pressmint-1.1/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>1</start>
<end>5</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T PressMint: Towards Interoperable Corpora of Historical Newspapers
%A Erjavec, Tomaž
%A Kopp, Matyáš
%A Ogrodniczuk, Maciej
%A Osenova, Petya
%A Rigau, German
%Y Ogrodniczuk, Maciej
%Y Osenova, Petya
%Y Wissik, Tanja
%S Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca, Spain
%F erjavec-etal-2026-pressmint
%X This paper presents Project X (name anonymized for review), an ongoing initiative to compile a multilingual, comparable, annotated, translated, and interoperable collection of European historical newspaper corpora. Spanning 17 countries and covering 15 languages, the project addresses a key shortcoming of existing newspaper resources: their lack of interoperability, which limits cross-lingual and transnational research. Building on the infrastructure and experience of the ParlaMint projects, the project adapts established encoding guidelines, validation workflows, and open-source tools to historical newspaper data. We outline the overall project architecture, the corpus encoding scheme, and the GitHub-based framework supporting collaborative development and quality control. The paper further describes the sample linguistic annotation pipeline, including OCR correction, text normalisation, and annotation within the Universal Dependencies framework, with attention to challenges posed by historical language varieties. The resulting FAIR, openly available corpora are intended to support comparative, diachronic research across the humanities and social sciences.
%R 10.63317/55w7588oukim
%U https://aclanthology.org/2026.pressmint-1.1/
%U https://doi.org/10.63317/55w7588oukim
%P 1-5
Markdown (Informal)
[PressMint: Towards Interoperable Corpora of Historical Newspapers](https://aclanthology.org/2026.pressmint-1.1/) (Erjavec et al., PressMint 2026)
ACL
- Tomaž Erjavec, Matyáš Kopp, Maciej Ogrodniczuk, Petya Osenova, and German Rigau. 2026. PressMint: Towards Interoperable Corpora of Historical Newspapers. In Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers, pages 1–5, Palma de Mallorca, Spain. Association for Computational Linguistics.