@inproceedings{abdelouahab-mohamed-2026-baflah,
title = "Baflah-lamri at {NAKBA}-{NLP} 2026: Manual Ground Truth Enrichment",
author = "Abdelouahab, Baflah and
Mohamed, Lamri",
editor = "Jarrar, Mustafa and
El-Haj, Mo and
Haddad, Amal and
Atiani, Serin and
Abudalfa, Shadi and
Regier, Terry and
Rayson, Paul and
Sima{'}an, Khalil and
Mansour, Camille",
booktitle = "Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.nakbanlp-1.16/",
doi = "10.63317/4aymwtp2ujku",
pages = "128--132",
abstract = "This paper presents a detailed description of the team{'}s methodology methodology in participating in Subtask 1 (Transcription Track) of the NAKBA NLP 2026 Shared Task for Arabic Manuscript Understanding. We present a rigorous approach to line-level manual transcription of historical Arabic manuscripts derived from the Omar Al-Saleh memoir collection (1951-1965). Our methodology emphatisez accuracy, consistency, and adherence to diplomatic transcription principles, while addressing the unique palaeographic and physical challenges of Arabic handwriting, such as writing speed, orthographic variation, and the impact of writing tools (e.g., immediate strike-throughs and ink spatter). The work guided by strict transcription guidelines and contextual verification protocols, matching cropped line images with full-page images to resolve ambiguities and automated cropping issues. The team successfully transcribed the entire assigned batch of 500 lines (100{\%} completion rate) across 368 unique pages, producing reference data comprising 6,719 words and 37,646 characters. This effort contributes to providing highly reliable Ground Truth data, serving as an essential foundation for training and evaluating Handwritten Text Recognition (HTR) models for Arabic manuscripts."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="abdelouahab-mohamed-2026-baflah">
<titleInfo>
<title>Baflah-lamri at NAKBA-NLP 2026: Manual Ground Truth Enrichment</title>
</titleInfo>
<name type="personal">
<namePart type="given">Baflah</namePart>
<namePart type="family">Abdelouahab</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Lamri</namePart>
<namePart type="family">Mohamed</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mustafa</namePart>
<namePart type="family">Jarrar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Amal</namePart>
<namePart type="family">Haddad</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Serin</namePart>
<namePart type="family">Atiani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shadi</namePart>
<namePart type="family">Abudalfa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Terry</namePart>
<namePart type="family">Regier</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paul</namePart>
<namePart type="family">Rayson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Khalil</namePart>
<namePart type="family">Sima’an</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Camille</namePart>
<namePart type="family">Mansour</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper presents a detailed description of the team’s methodology methodology in participating in Subtask 1 (Transcription Track) of the NAKBA NLP 2026 Shared Task for Arabic Manuscript Understanding. We present a rigorous approach to line-level manual transcription of historical Arabic manuscripts derived from the Omar Al-Saleh memoir collection (1951-1965). Our methodology emphatisez accuracy, consistency, and adherence to diplomatic transcription principles, while addressing the unique palaeographic and physical challenges of Arabic handwriting, such as writing speed, orthographic variation, and the impact of writing tools (e.g., immediate strike-throughs and ink spatter). The work guided by strict transcription guidelines and contextual verification protocols, matching cropped line images with full-page images to resolve ambiguities and automated cropping issues. The team successfully transcribed the entire assigned batch of 500 lines (100% completion rate) across 368 unique pages, producing reference data comprising 6,719 words and 37,646 characters. This effort contributes to providing highly reliable Ground Truth data, serving as an essential foundation for training and evaluating Handwritten Text Recognition (HTR) models for Arabic manuscripts.</abstract>
<identifier type="citekey">abdelouahab-mohamed-2026-baflah</identifier>
<identifier type="doi">10.63317/4aymwtp2ujku</identifier>
<location>
<url>https://aclanthology.org/2026.nakbanlp-1.16/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>128</start>
<end>132</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Baflah-lamri at NAKBA-NLP 2026: Manual Ground Truth Enrichment
%A Abdelouahab, Baflah
%A Mohamed, Lamri
%Y Jarrar, Mustafa
%Y El-Haj, Mo
%Y Haddad, Amal
%Y Atiani, Serin
%Y Abudalfa, Shadi
%Y Regier, Terry
%Y Rayson, Paul
%Y Sima’an, Khalil
%Y Mansour, Camille
%S Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F abdelouahab-mohamed-2026-baflah
%X This paper presents a detailed description of the team’s methodology methodology in participating in Subtask 1 (Transcription Track) of the NAKBA NLP 2026 Shared Task for Arabic Manuscript Understanding. We present a rigorous approach to line-level manual transcription of historical Arabic manuscripts derived from the Omar Al-Saleh memoir collection (1951-1965). Our methodology emphatisez accuracy, consistency, and adherence to diplomatic transcription principles, while addressing the unique palaeographic and physical challenges of Arabic handwriting, such as writing speed, orthographic variation, and the impact of writing tools (e.g., immediate strike-throughs and ink spatter). The work guided by strict transcription guidelines and contextual verification protocols, matching cropped line images with full-page images to resolve ambiguities and automated cropping issues. The team successfully transcribed the entire assigned batch of 500 lines (100% completion rate) across 368 unique pages, producing reference data comprising 6,719 words and 37,646 characters. This effort contributes to providing highly reliable Ground Truth data, serving as an essential foundation for training and evaluating Handwritten Text Recognition (HTR) models for Arabic manuscripts.
%R 10.63317/4aymwtp2ujku
%U https://aclanthology.org/2026.nakbanlp-1.16/
%U https://doi.org/10.63317/4aymwtp2ujku
%P 128-132
Markdown (Informal)
[Baflah-lamri at NAKBA-NLP 2026: Manual Ground Truth Enrichment](https://aclanthology.org/2026.nakbanlp-1.16/) (Abdelouahab & Mohamed, NakbaNLP 2026)
ACL