@inproceedings{abdo-etal-2026-tarikhi,
title = "Tarikhi: {A}rabic Temporal Information Extraction from {A}rabic Historical Documents",
author = "Abdo, Qusay and
Atiani, Serin and
Sraiji, Tariq and
Saeed, Adnan",
editor = "Jarrar, Mustafa and
El-Haj, Mo and
Haddad, Amal and
Atiani, Serin and
Abudalfa, Shadi and
Regier, Terry and
Rayson, Paul and
Sima{'}an, Khalil and
Mansour, Camille",
booktitle = "Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.nakbanlp-1.6/",
doi = "10.63317/5eco9hgcoxp8",
pages = "60--69",
abstract = "Arabic historical books and archival materials contain rich accounts of political, social, and cultural events, yet they remain largely underutilized computationally due to the scarcity of dedicated Arabic information extraction tools. The challenge is amplified in long-form, scanned historical documents, where optical character recognition noise, orthographic variation, and complex narrative structures complicate automatic processing. In this paper, we present Tarikhi, a retrieval-augmented generation framework for structured temporal event extraction from Arabic scanned books. The proposed pipeline integrates high-accuracy optical character recognition, chunking-based processing for long-document handling, Arabic named entity recognition, span refinement, and a retrieval-enhanced attribute extraction module that identifies event dates, locations, and descriptive summaries. Extracted events are consolidated and linked using semantic and temporal similarity measures, and linked through relation classification to construct structured temporal events. Evaluation on a selected part of modern Arabic historical books demonstrates the feasibility of temporal event extraction from long-form Arabic texts, achieving a 75.3{\%} F1-score under dual human verification. Tarikhi represents a step toward scalable temporal knowledge construction for Arabic digital humanities resources."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="abdo-etal-2026-tarikhi">
<titleInfo>
<title>Tarikhi: Arabic Temporal Information Extraction from Arabic Historical Documents</title>
</titleInfo>
<name type="personal">
<namePart type="given">Qusay</namePart>
<namePart type="family">Abdo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Serin</namePart>
<namePart type="family">Atiani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tariq</namePart>
<namePart type="family">Sraiji</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Adnan</namePart>
<namePart type="family">Saeed</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mustafa</namePart>
<namePart type="family">Jarrar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Amal</namePart>
<namePart type="family">Haddad</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Serin</namePart>
<namePart type="family">Atiani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shadi</namePart>
<namePart type="family">Abudalfa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Terry</namePart>
<namePart type="family">Regier</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paul</namePart>
<namePart type="family">Rayson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Khalil</namePart>
<namePart type="family">Sima’an</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Camille</namePart>
<namePart type="family">Mansour</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Arabic historical books and archival materials contain rich accounts of political, social, and cultural events, yet they remain largely underutilized computationally due to the scarcity of dedicated Arabic information extraction tools. The challenge is amplified in long-form, scanned historical documents, where optical character recognition noise, orthographic variation, and complex narrative structures complicate automatic processing. In this paper, we present Tarikhi, a retrieval-augmented generation framework for structured temporal event extraction from Arabic scanned books. The proposed pipeline integrates high-accuracy optical character recognition, chunking-based processing for long-document handling, Arabic named entity recognition, span refinement, and a retrieval-enhanced attribute extraction module that identifies event dates, locations, and descriptive summaries. Extracted events are consolidated and linked using semantic and temporal similarity measures, and linked through relation classification to construct structured temporal events. Evaluation on a selected part of modern Arabic historical books demonstrates the feasibility of temporal event extraction from long-form Arabic texts, achieving a 75.3% F1-score under dual human verification. Tarikhi represents a step toward scalable temporal knowledge construction for Arabic digital humanities resources.</abstract>
<identifier type="citekey">abdo-etal-2026-tarikhi</identifier>
<identifier type="doi">10.63317/5eco9hgcoxp8</identifier>
<location>
<url>https://aclanthology.org/2026.nakbanlp-1.6/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>60</start>
<end>69</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Tarikhi: Arabic Temporal Information Extraction from Arabic Historical Documents
%A Abdo, Qusay
%A Atiani, Serin
%A Sraiji, Tariq
%A Saeed, Adnan
%Y Jarrar, Mustafa
%Y El-Haj, Mo
%Y Haddad, Amal
%Y Atiani, Serin
%Y Abudalfa, Shadi
%Y Regier, Terry
%Y Rayson, Paul
%Y Sima’an, Khalil
%Y Mansour, Camille
%S Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F abdo-etal-2026-tarikhi
%X Arabic historical books and archival materials contain rich accounts of political, social, and cultural events, yet they remain largely underutilized computationally due to the scarcity of dedicated Arabic information extraction tools. The challenge is amplified in long-form, scanned historical documents, where optical character recognition noise, orthographic variation, and complex narrative structures complicate automatic processing. In this paper, we present Tarikhi, a retrieval-augmented generation framework for structured temporal event extraction from Arabic scanned books. The proposed pipeline integrates high-accuracy optical character recognition, chunking-based processing for long-document handling, Arabic named entity recognition, span refinement, and a retrieval-enhanced attribute extraction module that identifies event dates, locations, and descriptive summaries. Extracted events are consolidated and linked using semantic and temporal similarity measures, and linked through relation classification to construct structured temporal events. Evaluation on a selected part of modern Arabic historical books demonstrates the feasibility of temporal event extraction from long-form Arabic texts, achieving a 75.3% F1-score under dual human verification. Tarikhi represents a step toward scalable temporal knowledge construction for Arabic digital humanities resources.
%R 10.63317/5eco9hgcoxp8
%U https://aclanthology.org/2026.nakbanlp-1.6/
%U https://doi.org/10.63317/5eco9hgcoxp8
%P 60-69
Markdown (Informal)
[Tarikhi: Arabic Temporal Information Extraction from Arabic Historical Documents](https://aclanthology.org/2026.nakbanlp-1.6/) (Abdo et al., NakbaNLP 2026)
ACL