@inproceedings{aires-mendes-2026-pressmint,
title = "{P}ress{M}int-{PT} - Compiling a {P}ortuguese Historical Newspaper Corpus",
author = "Aires, Jose and
Mendes, Am{\'a}lia",
editor = "Ogrodniczuk, Maciej and
Osenova, Petya and
Wissik, Tanja",
booktitle = "Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.pressmint-1.4/",
doi = "10.63317/2os9smose6kj",
pages = "16--20",
abstract = "We present a new European Portuguese corpus of newspapers from the 19th and early 20th centuries, integrated in the recent PressMint project, whose goal is to provide a set of comparable newspaper corpora for European languages in that time frame. We discuss the raw data that was previously available, as well as new data specifically compiled for the project, and the challenges involving OCR, text recognition, and different orthographical norms. We describe the pipeline setup for XML encoding and annotation, partially based on work developed for the ParlaMint corpora. The corpus is currently under development and will be made freely available at the end of the project, as part of the PressMint corpora."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="aires-mendes-2026-pressmint">
<titleInfo>
<title>PressMint-PT - Compiling a Portuguese Historical Newspaper Corpus</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jose</namePart>
<namePart type="family">Aires</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Amália</namePart>
<namePart type="family">Mendes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Maciej</namePart>
<namePart type="family">Ogrodniczuk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Petya</namePart>
<namePart type="family">Osenova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tanja</namePart>
<namePart type="family">Wissik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present a new European Portuguese corpus of newspapers from the 19th and early 20th centuries, integrated in the recent PressMint project, whose goal is to provide a set of comparable newspaper corpora for European languages in that time frame. We discuss the raw data that was previously available, as well as new data specifically compiled for the project, and the challenges involving OCR, text recognition, and different orthographical norms. We describe the pipeline setup for XML encoding and annotation, partially based on work developed for the ParlaMint corpora. The corpus is currently under development and will be made freely available at the end of the project, as part of the PressMint corpora.</abstract>
<identifier type="citekey">aires-mendes-2026-pressmint</identifier>
<identifier type="doi">10.63317/2os9smose6kj</identifier>
<location>
<url>https://aclanthology.org/2026.pressmint-1.4/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>16</start>
<end>20</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T PressMint-PT - Compiling a Portuguese Historical Newspaper Corpus
%A Aires, Jose
%A Mendes, Amália
%Y Ogrodniczuk, Maciej
%Y Osenova, Petya
%Y Wissik, Tanja
%S Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca, Spain
%F aires-mendes-2026-pressmint
%X We present a new European Portuguese corpus of newspapers from the 19th and early 20th centuries, integrated in the recent PressMint project, whose goal is to provide a set of comparable newspaper corpora for European languages in that time frame. We discuss the raw data that was previously available, as well as new data specifically compiled for the project, and the challenges involving OCR, text recognition, and different orthographical norms. We describe the pipeline setup for XML encoding and annotation, partially based on work developed for the ParlaMint corpora. The corpus is currently under development and will be made freely available at the end of the project, as part of the PressMint corpora.
%R 10.63317/2os9smose6kj
%U https://aclanthology.org/2026.pressmint-1.4/
%U https://doi.org/10.63317/2os9smose6kj
%P 16-20
Markdown (Informal)
[PressMint-PT - Compiling a Portuguese Historical Newspaper Corpus](https://aclanthology.org/2026.pressmint-1.4/) (Aires & Mendes, PressMint 2026)
ACL