@inproceedings{nawaz-etal-2026-towards,
title = "Towards Benchmarking {O}ld {C}hurch {S}lavonic Lemmatization",
author = "Nawaz, Usman and
Napolitano, Marianna and
Karafillidis, Iris and
Lo Presti, Liliana and
Cascia, Marco",
editor = "Bernard, Timoth{\'e}e and
Chersoni, Emmanuele and
Rambelli, Giulia",
booktitle = "Proceedings of the Third Workshop on the Bridges and Gaps between Formal and Computational Linguistics ({B}ri{G}ap-3)",
month = jul,
year = "2026",
address = "Paris, France",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.brigap-1.9/",
pages = "82--93",
abstract = "Lemmatization is an important preprocessing step in Natural Language Processing (NLP); however, annotated resources for medieval languages such as Old Church Slavonic (OCS) are limited in scope, size, and diversity. This paper presents the annotated resources for OCS lemmatization, including annotation process, design choices and non-standard Unicode related issues. The annotated corpus is used to evaluate existing lemmatization tools (Stanza and UDPipe-2 models trained on the UD 2.12 treebank, and a dictionary-based approach) both in cross-dataset and on a corpus obtained by merging the new annotations with existing UD V2.12 OCS data. Pretrained models perform poorly ({\ensuremath{\approx}} 15{--}16{\%}), below a dictionary baseline ({\ensuremath{\approx}} 38{\%}), while retraining on the new data improves performance (up to {\ensuremath{\approx}} 51{\%}) and shows different cross-dataset generalization. Experiments in cross-dataset and on the combined corpus demonstrate that lemmatization performance depends strongly on dataset similarity, annotation conventions, and orthographic mismatch. Overall, the findings show the value of the newly annotated resources and the importance of extending OCS lemmatization benchmarks for historical Slavic NLP."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="nawaz-etal-2026-towards">
<titleInfo>
<title>Towards Benchmarking Old Church Slavonic Lemmatization</title>
</titleInfo>
<name type="personal">
<namePart type="given">Usman</namePart>
<namePart type="family">Nawaz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marianna</namePart>
<namePart type="family">Napolitano</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Iris</namePart>
<namePart type="family">Karafillidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Liliana</namePart>
<namePart type="family">Lo Presti</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marco</namePart>
<namePart type="family">Cascia</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-07</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Third Workshop on the Bridges and Gaps between Formal and Computational Linguistics (BriGap-3)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Timothée</namePart>
<namePart type="family">Bernard</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Emmanuele</namePart>
<namePart type="family">Chersoni</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Giulia</namePart>
<namePart type="family">Rambelli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Paris, France</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Lemmatization is an important preprocessing step in Natural Language Processing (NLP); however, annotated resources for medieval languages such as Old Church Slavonic (OCS) are limited in scope, size, and diversity. This paper presents the annotated resources for OCS lemmatization, including annotation process, design choices and non-standard Unicode related issues. The annotated corpus is used to evaluate existing lemmatization tools (Stanza and UDPipe-2 models trained on the UD 2.12 treebank, and a dictionary-based approach) both in cross-dataset and on a corpus obtained by merging the new annotations with existing UD V2.12 OCS data. Pretrained models perform poorly (\ensuremath\approx 15–16%), below a dictionary baseline (\ensuremath\approx 38%), while retraining on the new data improves performance (up to \ensuremath\approx 51%) and shows different cross-dataset generalization. Experiments in cross-dataset and on the combined corpus demonstrate that lemmatization performance depends strongly on dataset similarity, annotation conventions, and orthographic mismatch. Overall, the findings show the value of the newly annotated resources and the importance of extending OCS lemmatization benchmarks for historical Slavic NLP.</abstract>
<identifier type="citekey">nawaz-etal-2026-towards</identifier>
<location>
<url>https://aclanthology.org/2026.brigap-1.9/</url>
</location>
<part>
<date>2026-07</date>
<extent unit="page">
<start>82</start>
<end>93</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Towards Benchmarking Old Church Slavonic Lemmatization
%A Nawaz, Usman
%A Napolitano, Marianna
%A Karafillidis, Iris
%A Lo Presti, Liliana
%A Cascia, Marco
%Y Bernard, Timothée
%Y Chersoni, Emmanuele
%Y Rambelli, Giulia
%S Proceedings of the Third Workshop on the Bridges and Gaps between Formal and Computational Linguistics (BriGap-3)
%D 2026
%8 July
%I Association for Computational Linguistics
%C Paris, France
%F nawaz-etal-2026-towards
%X Lemmatization is an important preprocessing step in Natural Language Processing (NLP); however, annotated resources for medieval languages such as Old Church Slavonic (OCS) are limited in scope, size, and diversity. This paper presents the annotated resources for OCS lemmatization, including annotation process, design choices and non-standard Unicode related issues. The annotated corpus is used to evaluate existing lemmatization tools (Stanza and UDPipe-2 models trained on the UD 2.12 treebank, and a dictionary-based approach) both in cross-dataset and on a corpus obtained by merging the new annotations with existing UD V2.12 OCS data. Pretrained models perform poorly (\ensuremath\approx 15–16%), below a dictionary baseline (\ensuremath\approx 38%), while retraining on the new data improves performance (up to \ensuremath\approx 51%) and shows different cross-dataset generalization. Experiments in cross-dataset and on the combined corpus demonstrate that lemmatization performance depends strongly on dataset similarity, annotation conventions, and orthographic mismatch. Overall, the findings show the value of the newly annotated resources and the importance of extending OCS lemmatization benchmarks for historical Slavic NLP.
%U https://aclanthology.org/2026.brigap-1.9/
%P 82-93
Markdown (Informal)
[Towards Benchmarking Old Church Slavonic Lemmatization](https://aclanthology.org/2026.brigap-1.9/) (Nawaz et al., BriGap 2026)
ACL
- Usman Nawaz, Marianna Napolitano, Iris Karafillidis, Liliana Lo Presti, and Marco Cascia. 2026. Towards Benchmarking Old Church Slavonic Lemmatization. In Proceedings of the Third Workshop on the Bridges and Gaps between Formal and Computational Linguistics (BriGap-3), pages 82–93, Paris, France. Association for Computational Linguistics.