@inproceedings{paev-etal-2026-towards,
title = "Towards a {B}ulgarian Historical Newspaper Corpus {--} Construction of Reading Order over the Text in Searchable {PDF}s",
author = "Paev, Nikolay and
Marinov, Stefan and
Kratchanov, Ivan and
Osenova, Petya and
Simov, Kiril",
editor = "Ogrodniczuk, Maciej and
Osenova, Petya and
Wissik, Tanja",
booktitle = "Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.pressmint-1.10/",
doi = "10.63317/336sd7ixv3fw",
pages = "56--64",
abstract = "The determine the reading order of the text extracted from a searchable PDF produced by an OCR software from an old newspaper is the first task in the process of preparation of corpora of old newspapers. In the paper we present an algorithm for generation of reading order of black selected from the corresponding PDF. Also we performed a tuning of the parameters of the algorithm. The optimization provides 10 {\%} improvement."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="paev-etal-2026-towards">
<titleInfo>
<title>Towards a Bulgarian Historical Newspaper Corpus – Construction of Reading Order over the Text in Searchable PDFs</title>
</titleInfo>
<name type="personal">
<namePart type="given">Nikolay</namePart>
<namePart type="family">Paev</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stefan</namePart>
<namePart type="family">Marinov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ivan</namePart>
<namePart type="family">Kratchanov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Petya</namePart>
<namePart type="family">Osenova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kiril</namePart>
<namePart type="family">Simov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers</title>
</titleInfo>
<name type="personal">
<namePart type="given">Maciej</namePart>
<namePart type="family">Ogrodniczuk</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Petya</namePart>
<namePart type="family">Osenova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tanja</namePart>
<namePart type="family">Wissik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The determine the reading order of the text extracted from a searchable PDF produced by an OCR software from an old newspaper is the first task in the process of preparation of corpora of old newspapers. In the paper we present an algorithm for generation of reading order of black selected from the corresponding PDF. Also we performed a tuning of the parameters of the algorithm. The optimization provides 10 % improvement.</abstract>
<identifier type="citekey">paev-etal-2026-towards</identifier>
<identifier type="doi">10.63317/336sd7ixv3fw</identifier>
<location>
<url>https://aclanthology.org/2026.pressmint-1.10/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>56</start>
<end>64</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Towards a Bulgarian Historical Newspaper Corpus – Construction of Reading Order over the Text in Searchable PDFs
%A Paev, Nikolay
%A Marinov, Stefan
%A Kratchanov, Ivan
%A Osenova, Petya
%A Simov, Kiril
%Y Ogrodniczuk, Maciej
%Y Osenova, Petya
%Y Wissik, Tanja
%S Proceedings of the First Workshop on Creating Interoperable Corpora of Historical Newspapers
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma de Mallorca, Spain
%F paev-etal-2026-towards
%X The determine the reading order of the text extracted from a searchable PDF produced by an OCR software from an old newspaper is the first task in the process of preparation of corpora of old newspapers. In the paper we present an algorithm for generation of reading order of black selected from the corresponding PDF. Also we performed a tuning of the parameters of the algorithm. The optimization provides 10 % improvement.
%R 10.63317/336sd7ixv3fw
%U https://aclanthology.org/2026.pressmint-1.10/
%U https://doi.org/10.63317/336sd7ixv3fw
%P 56-64
Markdown (Informal)
[Towards a Bulgarian Historical Newspaper Corpus – Construction of Reading Order over the Text in Searchable PDFs](https://aclanthology.org/2026.pressmint-1.10/) (Paev et al., PressMint 2026)
ACL