@inproceedings{corbetta-etal-2026-beyond,
title = "Beyond {OCR}: Structural Segmentation and Speaker Attribution in Historical {I}talian Parliamentary Debates",
author = "Corbetta, Claudia and
Mazzei, Samuele and
Palmero Aprosio, Alessio",
editor = "Eskevich, Maria and
Vandeghinste, Vincent and
Bodron, David",
booktitle = "Proceedings of the {P}arla{CLARIN} {V} Workshop on Interoperability, Multilinguality, and Multimodality in Parliamentary Corpora",
month = may,
year = "2026",
address = "Palma de Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.parlaclarin-1.8/",
doi = "10.63317/39yi5mff3w3j",
pages = "65--76",
abstract = "Historical parliamentary debates are essential for longitudinal political and linguistic research, yet much early material remains available only as scanned images. In the Italian context, proceedings from 1848{--}1996 lack large-scale, structurally annotated, machine-readable representations. This paper addresses the challenge of transforming historical Italian parliamentary debates into structured corpora by moving beyond plain Optical Character Recognition (OCR) toward functional block segmentation and speaker attribution. We present detailed annotation guidelines and a manually annotated dataset of 300 randomly sampled pages. Two approaches are compared: (i) direct multimodal Large Language Model (LLM) annotation and (ii) a modular pipeline combining OCR with LLM-based structural reconstruction under zero-shot and few-shot prompting. Evaluation on a held-out test set shows that separating transcription from structural reasoning improves performance, with few-shot prompting yielding the most reliable results. The study demonstrates the feasibility of integrating LLM-based reasoning into historical parliamentary digitisation workflows."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="corbetta-etal-2026-beyond">
<titleInfo>
<title>Beyond OCR: Structural Segmentation and Speaker Attribution in Historical Italian Parliamentary Debates</title>
</titleInfo>
<name type="personal">
<namePart type="given">Claudia</namePart>
<namePart type="family">Corbetta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Samuele</namePart>
<namePart type="family">Mazzei</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alessio</namePart>
<namePart type="family">Palmero Aprosio</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the ParlaCLARIN V Workshop on Interoperability, Multilinguality, and Multimodality in Parliamentary Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Maria</namePart>
<namePart type="family">Eskevich</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vincent</namePart>
<namePart type="family">Vandeghinste</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">David</namePart>
<namePart type="family">Bodron</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Historical parliamentary debates are essential for longitudinal political and linguistic research, yet much early material remains available only as scanned images. In the Italian context, proceedings from 1848–1996 lack large-scale, structurally annotated, machine-readable representations. This paper addresses the challenge of transforming historical Italian parliamentary debates into structured corpora by moving beyond plain Optical Character Recognition (OCR) toward functional block segmentation and speaker attribution. We present detailed annotation guidelines and a manually annotated dataset of 300 randomly sampled pages. Two approaches are compared: (i) direct multimodal Large Language Model (LLM) annotation and (ii) a modular pipeline combining OCR with LLM-based structural reconstruction under zero-shot and few-shot prompting. Evaluation on a held-out test set shows that separating transcription from structural reasoning improves performance, with few-shot prompting yielding the most reliable results. The study demonstrates the feasibility of integrating LLM-based reasoning into historical parliamentary digitisation workflows.</abstract>
<identifier type="citekey">corbetta-etal-2026-beyond</identifier>
<identifier type="doi">10.63317/39yi5mff3w3j</identifier>
<location>
<url>https://aclanthology.org/2026.parlaclarin-1.8/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>65</start>
<end>76</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Beyond OCR: Structural Segmentation and Speaker Attribution in Historical Italian Parliamentary Debates
%A Corbetta, Claudia
%A Mazzei, Samuele
%A Palmero Aprosio, Alessio
%Y Eskevich, Maria
%Y Vandeghinste, Vincent
%Y Bodron, David
%S Proceedings of the ParlaCLARIN V Workshop on Interoperability, Multilinguality, and Multimodality in Parliamentary Corpora
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca (Spain)
%F corbetta-etal-2026-beyond
%X Historical parliamentary debates are essential for longitudinal political and linguistic research, yet much early material remains available only as scanned images. In the Italian context, proceedings from 1848–1996 lack large-scale, structurally annotated, machine-readable representations. This paper addresses the challenge of transforming historical Italian parliamentary debates into structured corpora by moving beyond plain Optical Character Recognition (OCR) toward functional block segmentation and speaker attribution. We present detailed annotation guidelines and a manually annotated dataset of 300 randomly sampled pages. Two approaches are compared: (i) direct multimodal Large Language Model (LLM) annotation and (ii) a modular pipeline combining OCR with LLM-based structural reconstruction under zero-shot and few-shot prompting. Evaluation on a held-out test set shows that separating transcription from structural reasoning improves performance, with few-shot prompting yielding the most reliable results. The study demonstrates the feasibility of integrating LLM-based reasoning into historical parliamentary digitisation workflows.
%R 10.63317/39yi5mff3w3j
%U https://aclanthology.org/2026.parlaclarin-1.8/
%U https://doi.org/10.63317/39yi5mff3w3j
%P 65-76
Markdown (Informal)
[Beyond OCR: Structural Segmentation and Speaker Attribution in Historical Italian Parliamentary Debates](https://aclanthology.org/2026.parlaclarin-1.8/) (Corbetta et al., ParlaCLARIN 2026)
ACL