@inproceedings{giouli-etal-2026-calibrated,
title = "A Calibrated and Interpretable Framework for Multilingual Text Difficulty Prediction",
author = "Giouli, Voula and
Tsoulouhas, George and
Sioupi, Athina and
Michalopoulou, Stamatia",
editor = "Di Nunzio, Giorgio Maria and
Vezzani, Federica and
Ermakova, Liana and
Azarbonyad, Hosein and
Kamps, Jaap",
booktitle = "Proceedings of the 2nd Workshop on Evaluating Text Difficulty in a Multilingual Context ({D}e{T}erm{I}t! 2026)",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.determit-1.7/",
doi = "10.63317/3cmykny29qkb",
pages = "63--73",
abstract = "Text difficulty prediction in educational contexts requires models that balance predictive performance, interpretability, calibration, and pedagogical alignment. While transformer-based approaches increasingly dominate text difficulty classification, educational applications demand transparent and linguistically grounded modeling. This paper presents work aimed at developing a workbench for CEFR-based text difficulty prediction. The proposed platform comprises three main components: (i) a tool for CEFR-aligned dataset preparation incorporating a pipeline for documenting, processing, and enriching textual data, (ii) CEFR-aligned datasets, and (iii) three alternative modeling approaches, namely a rule-based baseline, a feature-based Machine Learning (ML) classifier, and a fine-tuned BERT model. Our approach integrates linguistically informed feature engineering with data-driven modeling techniques, thereby balancing transparency and predictive performance. The proposed workbench has been designed as a language-agnostic infrastructure that can be extended to any language. In its current implementation, it has been applied to the creation of a German CEFR dataset, while its Greek counterpart is currently under development."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="giouli-etal-2026-calibrated">
<titleInfo>
<title>A Calibrated and Interpretable Framework for Multilingual Text Difficulty Prediction</title>
</titleInfo>
<name type="personal">
<namePart type="given">Voula</namePart>
<namePart type="family">Giouli</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">George</namePart>
<namePart type="family">Tsoulouhas</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Athina</namePart>
<namePart type="family">Sioupi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Stamatia</namePart>
<namePart type="family">Michalopoulou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 2nd Workshop on Evaluating Text Difficulty in a Multilingual Context (DeTermIt! 2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Giorgio</namePart>
<namePart type="given">Maria</namePart>
<namePart type="family">Di Nunzio</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Federica</namePart>
<namePart type="family">Vezzani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Liana</namePart>
<namePart type="family">Ermakova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hosein</namePart>
<namePart type="family">Azarbonyad</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jaap</namePart>
<namePart type="family">Kamps</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Text difficulty prediction in educational contexts requires models that balance predictive performance, interpretability, calibration, and pedagogical alignment. While transformer-based approaches increasingly dominate text difficulty classification, educational applications demand transparent and linguistically grounded modeling. This paper presents work aimed at developing a workbench for CEFR-based text difficulty prediction. The proposed platform comprises three main components: (i) a tool for CEFR-aligned dataset preparation incorporating a pipeline for documenting, processing, and enriching textual data, (ii) CEFR-aligned datasets, and (iii) three alternative modeling approaches, namely a rule-based baseline, a feature-based Machine Learning (ML) classifier, and a fine-tuned BERT model. Our approach integrates linguistically informed feature engineering with data-driven modeling techniques, thereby balancing transparency and predictive performance. The proposed workbench has been designed as a language-agnostic infrastructure that can be extended to any language. In its current implementation, it has been applied to the creation of a German CEFR dataset, while its Greek counterpart is currently under development.</abstract>
<identifier type="citekey">giouli-etal-2026-calibrated</identifier>
<identifier type="doi">10.63317/3cmykny29qkb</identifier>
<location>
<url>https://aclanthology.org/2026.determit-1.7/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>63</start>
<end>73</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T A Calibrated and Interpretable Framework for Multilingual Text Difficulty Prediction
%A Giouli, Voula
%A Tsoulouhas, George
%A Sioupi, Athina
%A Michalopoulou, Stamatia
%Y Di Nunzio, Giorgio Maria
%Y Vezzani, Federica
%Y Ermakova, Liana
%Y Azarbonyad, Hosein
%Y Kamps, Jaap
%S Proceedings of the 2nd Workshop on Evaluating Text Difficulty in a Multilingual Context (DeTermIt! 2026)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F giouli-etal-2026-calibrated
%X Text difficulty prediction in educational contexts requires models that balance predictive performance, interpretability, calibration, and pedagogical alignment. While transformer-based approaches increasingly dominate text difficulty classification, educational applications demand transparent and linguistically grounded modeling. This paper presents work aimed at developing a workbench for CEFR-based text difficulty prediction. The proposed platform comprises three main components: (i) a tool for CEFR-aligned dataset preparation incorporating a pipeline for documenting, processing, and enriching textual data, (ii) CEFR-aligned datasets, and (iii) three alternative modeling approaches, namely a rule-based baseline, a feature-based Machine Learning (ML) classifier, and a fine-tuned BERT model. Our approach integrates linguistically informed feature engineering with data-driven modeling techniques, thereby balancing transparency and predictive performance. The proposed workbench has been designed as a language-agnostic infrastructure that can be extended to any language. In its current implementation, it has been applied to the creation of a German CEFR dataset, while its Greek counterpart is currently under development.
%R 10.63317/3cmykny29qkb
%U https://aclanthology.org/2026.determit-1.7/
%U https://doi.org/10.63317/3cmykny29qkb
%P 63-73
Markdown (Informal)
[A Calibrated and Interpretable Framework for Multilingual Text Difficulty Prediction](https://aclanthology.org/2026.determit-1.7/) (Giouli et al., DeTermIt 2026)
ACL