@inproceedings{ciletti-2026-foggia,
title = "The {F}oggia Occupator Corpus: Digitisation, Annotation, and Computational Analysis of an Occupation-Era Newspaper (1945-1946)",
author = "Ciletti, Michele",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.536/",
doi = "10.63317/2xbscmzsjers",
pages = "6731--6739",
abstract = "Historical newspapers are crucial sources yet often remain undigitised or lack machine-readable text. We present the Foggia Occupator corpus, a linguistically enriched, openly licensed resource built from twenty-two issues (Dec 1945{--}Aug 1946) of a weekly newspaper produced by U.S. personnel in occupied Foggia, Italy. High-resolution scans were processed via OCR with LLM-assisted correction (GPT-4o) and full human verification, then segmented into 874 articles ( 216k tokens). We annotate topics, named entities and typed relations via a semi-automatic pipeline with manual reconciliation, and perform argument mining on civics- and conflict-related content, yielding 1,735 arguments. The entity{--}relation layer supports network analyses that reveal sparse, modular structures linking military units, civic bodies, and social life. We release TEI-XML with entity spans, JSON article files with metadata, CSVs of entities/relations with temporal counts, and an arguments JSON, all under a Creative Commons 4.0 licence. Beyond documenting an in-between moment of reconstruction, the resource enables benchmarking for OCR-robust NER/RE and studies of framing, stance, and community structure in post-war local media."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="ciletti-2026-foggia">
<titleInfo>
<title>The Foggia Occupator Corpus: Digitisation, Annotation, and Computational Analysis of an Occupation-Era Newspaper (1945-1946)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Michele</namePart>
<namePart type="family">Ciletti</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Historical newspapers are crucial sources yet often remain undigitised or lack machine-readable text. We present the Foggia Occupator corpus, a linguistically enriched, openly licensed resource built from twenty-two issues (Dec 1945–Aug 1946) of a weekly newspaper produced by U.S. personnel in occupied Foggia, Italy. High-resolution scans were processed via OCR with LLM-assisted correction (GPT-4o) and full human verification, then segmented into 874 articles ( 216k tokens). We annotate topics, named entities and typed relations via a semi-automatic pipeline with manual reconciliation, and perform argument mining on civics- and conflict-related content, yielding 1,735 arguments. The entity–relation layer supports network analyses that reveal sparse, modular structures linking military units, civic bodies, and social life. We release TEI-XML with entity spans, JSON article files with metadata, CSVs of entities/relations with temporal counts, and an arguments JSON, all under a Creative Commons 4.0 licence. Beyond documenting an in-between moment of reconstruction, the resource enables benchmarking for OCR-robust NER/RE and studies of framing, stance, and community structure in post-war local media.</abstract>
<identifier type="citekey">ciletti-2026-foggia</identifier>
<identifier type="doi">10.63317/2xbscmzsjers</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.536/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>6731</start>
<end>6739</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T The Foggia Occupator Corpus: Digitisation, Annotation, and Computational Analysis of an Occupation-Era Newspaper (1945-1946)
%A Ciletti, Michele
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F ciletti-2026-foggia
%X Historical newspapers are crucial sources yet often remain undigitised or lack machine-readable text. We present the Foggia Occupator corpus, a linguistically enriched, openly licensed resource built from twenty-two issues (Dec 1945–Aug 1946) of a weekly newspaper produced by U.S. personnel in occupied Foggia, Italy. High-resolution scans were processed via OCR with LLM-assisted correction (GPT-4o) and full human verification, then segmented into 874 articles ( 216k tokens). We annotate topics, named entities and typed relations via a semi-automatic pipeline with manual reconciliation, and perform argument mining on civics- and conflict-related content, yielding 1,735 arguments. The entity–relation layer supports network analyses that reveal sparse, modular structures linking military units, civic bodies, and social life. We release TEI-XML with entity spans, JSON article files with metadata, CSVs of entities/relations with temporal counts, and an arguments JSON, all under a Creative Commons 4.0 licence. Beyond documenting an in-between moment of reconstruction, the resource enables benchmarking for OCR-robust NER/RE and studies of framing, stance, and community structure in post-war local media.
%R 10.63317/2xbscmzsjers
%U https://aclanthology.org/2026.lrec-1.536/
%U https://doi.org/10.63317/2xbscmzsjers
%P 6731-6739
Markdown (Informal)
[The Foggia Occupator Corpus: Digitisation, Annotation, and Computational Analysis of an Occupation-Era Newspaper (1945-1946)](https://aclanthology.org/2026.lrec-1.536/) (Ciletti, LREC 2026)
ACL