@inproceedings{nikolova-stoupak-etal-2026-automatic,
title = "Automatic Generation of Graded Texts in {O}ld {C}hurch {S}lavonic",
author = {Nikolova-Stoupak, Iglika and
Lejeune, Ga{\"e}l and
Shestakova-Stukun, Aliona and
Schaeffer-Lacroix, Eva},
editor = "Di Nunzio, Giorgio Maria and
Vezzani, Federica and
Ermakova, Liana and
Azarbonyad, Hosein and
Kamps, Jaap",
booktitle = "Proceedings of the 2nd Workshop on Evaluating Text Difficulty in a Multilingual Context ({D}e{T}erm{I}t! 2026)",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.determit-1.6/",
doi = "10.63317/3mgy6qi4bkoq",
pages = "51--62",
abstract = "In the past few decades, graded readers have been valued within language education and have so much as extended onto the so-called classical (or `dead') languages, such as Latin and Greek. The immersive reading and listening of adapted texts in these languages has been shown to increase students' proficiency, independence and motivation. However, as of now there is only a small number of related resources as well as of classical languages represented. The present study will investigate the current potential for (semi-)automatic generation of adapted classical-language readers while focusing on the Old Church Slavonic language. From a Natural Language Processing (NLP) point of view, work with the language is challenging due to the variety of dialects and diachronic variations it encompasses. The following steps are taken within our study: 1) Representative measurable characteristics of professional classical-language readers, such as the Latin Lingua latina per se illustrata and the Greek Athenaze, are analysed. 2) Automatic generation of adapted Old Church Slavonic text is attempted through the use of a sequence-to-sequence model (mT5) as well as a Large Language Model (GPT-5) in a one-shot setting. 3) The derived texts' quality is assessed through both human evaluation and a comparison of their textual characteristics with those of professional texts as defined in point 1). The edited versions of the GPT-based texts are shared for future reference and use."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="nikolova-stoupak-etal-2026-automatic">
<titleInfo>
<title>Automatic Generation of Graded Texts in Old Church Slavonic</title>
</titleInfo>
<name type="personal">
<namePart type="given">Iglika</namePart>
<namePart type="family">Nikolova-Stoupak</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gaël</namePart>
<namePart type="family">Lejeune</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aliona</namePart>
<namePart type="family">Shestakova-Stukun</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eva</namePart>
<namePart type="family">Schaeffer-Lacroix</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 2nd Workshop on Evaluating Text Difficulty in a Multilingual Context (DeTermIt! 2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Giorgio</namePart>
<namePart type="given">Maria</namePart>
<namePart type="family">Di Nunzio</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Federica</namePart>
<namePart type="family">Vezzani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Liana</namePart>
<namePart type="family">Ermakova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hosein</namePart>
<namePart type="family">Azarbonyad</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jaap</namePart>
<namePart type="family">Kamps</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>In the past few decades, graded readers have been valued within language education and have so much as extended onto the so-called classical (or ‘dead’) languages, such as Latin and Greek. The immersive reading and listening of adapted texts in these languages has been shown to increase students’ proficiency, independence and motivation. However, as of now there is only a small number of related resources as well as of classical languages represented. The present study will investigate the current potential for (semi-)automatic generation of adapted classical-language readers while focusing on the Old Church Slavonic language. From a Natural Language Processing (NLP) point of view, work with the language is challenging due to the variety of dialects and diachronic variations it encompasses. The following steps are taken within our study: 1) Representative measurable characteristics of professional classical-language readers, such as the Latin Lingua latina per se illustrata and the Greek Athenaze, are analysed. 2) Automatic generation of adapted Old Church Slavonic text is attempted through the use of a sequence-to-sequence model (mT5) as well as a Large Language Model (GPT-5) in a one-shot setting. 3) The derived texts’ quality is assessed through both human evaluation and a comparison of their textual characteristics with those of professional texts as defined in point 1). The edited versions of the GPT-based texts are shared for future reference and use.</abstract>
<identifier type="citekey">nikolova-stoupak-etal-2026-automatic</identifier>
<identifier type="doi">10.63317/3mgy6qi4bkoq</identifier>
<location>
<url>https://aclanthology.org/2026.determit-1.6/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>51</start>
<end>62</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Automatic Generation of Graded Texts in Old Church Slavonic
%A Nikolova-Stoupak, Iglika
%A Lejeune, Gaël
%A Shestakova-Stukun, Aliona
%A Schaeffer-Lacroix, Eva
%Y Di Nunzio, Giorgio Maria
%Y Vezzani, Federica
%Y Ermakova, Liana
%Y Azarbonyad, Hosein
%Y Kamps, Jaap
%S Proceedings of the 2nd Workshop on Evaluating Text Difficulty in a Multilingual Context (DeTermIt! 2026)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F nikolova-stoupak-etal-2026-automatic
%X In the past few decades, graded readers have been valued within language education and have so much as extended onto the so-called classical (or ‘dead’) languages, such as Latin and Greek. The immersive reading and listening of adapted texts in these languages has been shown to increase students’ proficiency, independence and motivation. However, as of now there is only a small number of related resources as well as of classical languages represented. The present study will investigate the current potential for (semi-)automatic generation of adapted classical-language readers while focusing on the Old Church Slavonic language. From a Natural Language Processing (NLP) point of view, work with the language is challenging due to the variety of dialects and diachronic variations it encompasses. The following steps are taken within our study: 1) Representative measurable characteristics of professional classical-language readers, such as the Latin Lingua latina per se illustrata and the Greek Athenaze, are analysed. 2) Automatic generation of adapted Old Church Slavonic text is attempted through the use of a sequence-to-sequence model (mT5) as well as a Large Language Model (GPT-5) in a one-shot setting. 3) The derived texts’ quality is assessed through both human evaluation and a comparison of their textual characteristics with those of professional texts as defined in point 1). The edited versions of the GPT-based texts are shared for future reference and use.
%R 10.63317/3mgy6qi4bkoq
%U https://aclanthology.org/2026.determit-1.6/
%U https://doi.org/10.63317/3mgy6qi4bkoq
%P 51-62
Markdown (Informal)
[Automatic Generation of Graded Texts in Old Church Slavonic](https://aclanthology.org/2026.determit-1.6/) (Nikolova-Stoupak et al., DeTermIt 2026)
ACL
- Iglika Nikolova-Stoupak, Gaël Lejeune, Aliona Shestakova-Stukun, and Eva Schaeffer-Lacroix. 2026. Automatic Generation of Graded Texts in Old Church Slavonic. In Proceedings of the 2nd Workshop on Evaluating Text Difficulty in a Multilingual Context (DeTermIt! 2026), pages 51–62, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).