@inproceedings{serras-etal-2026-compression,
title = "Compression-Based Linguistic Complexity Metrics in Automatic Essay Scoring",
author = "Serras, Felipe Ribas and
Silveira, Igor Cataneo and
Mau{\'a}, Denis Deratani and
Finger, Marcelo",
editor = "Barbosa, Bryan Khelven da Silva and
Paes, Aline and
Felippo, Ariani Di",
booktitle = "Proceedings of the 17th {B}razilian Symposium in Information and Human Language Technology",
month = oct,
year = "2026",
address = "Cuiab{\'a}, Mato Grosso, Brazil",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.stil-1.32/",
doi = "10.5753/stil.2026.26572",
pages = "390--404",
abstract = "Compression-based linguistic complexity metrics enable cross-linguistic comparison without prior annotation. Their sensitivity to variation across languages and Portuguese registers highlights their applicability in NLP tasks. This study investigates their use as readability proxies and complementary features in Automatic Essay Scoring. We analyze how these metrics capture variation in essay quality across traits, genres, and educational levels in Brazilian Portuguese. In addition, we evaluate their sensitivity to differences between humanand AI-generated essays. Our results suggest that complexity metrics are effective (i) in differentiating educational levels, (ii) in detecting whether they were written by humans and (iii) as predictors of essay quality."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="serras-etal-2026-compression">
<titleInfo>
<title>Compression-Based Linguistic Complexity Metrics in Automatic Essay Scoring</title>
</titleInfo>
<name type="personal">
<namePart type="given">Felipe</namePart>
<namePart type="given">Ribas</namePart>
<namePart type="family">Serras</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Igor</namePart>
<namePart type="given">Cataneo</namePart>
<namePart type="family">Silveira</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Denis</namePart>
<namePart type="given">Deratani</namePart>
<namePart type="family">Mauá</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marcelo</namePart>
<namePart type="family">Finger</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-10</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology</title>
</titleInfo>
<name type="personal">
<namePart type="given">Bryan</namePart>
<namePart type="given">Khelven</namePart>
<namePart type="given">da</namePart>
<namePart type="given">Silva</namePart>
<namePart type="family">Barbosa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Aline</namePart>
<namePart type="family">Paes</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ariani</namePart>
<namePart type="given">Di</namePart>
<namePart type="family">Felippo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Cuiabá, Mato Grosso, Brazil</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Compression-based linguistic complexity metrics enable cross-linguistic comparison without prior annotation. Their sensitivity to variation across languages and Portuguese registers highlights their applicability in NLP tasks. This study investigates their use as readability proxies and complementary features in Automatic Essay Scoring. We analyze how these metrics capture variation in essay quality across traits, genres, and educational levels in Brazilian Portuguese. In addition, we evaluate their sensitivity to differences between humanand AI-generated essays. Our results suggest that complexity metrics are effective (i) in differentiating educational levels, (ii) in detecting whether they were written by humans and (iii) as predictors of essay quality.</abstract>
<identifier type="citekey">serras-etal-2026-compression</identifier>
<identifier type="doi">10.5753/stil.2026.26572</identifier>
<location>
<url>https://aclanthology.org/2026.stil-1.32/</url>
</location>
<part>
<date>2026-10</date>
<extent unit="page">
<start>390</start>
<end>404</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Compression-Based Linguistic Complexity Metrics in Automatic Essay Scoring
%A Serras, Felipe Ribas
%A Silveira, Igor Cataneo
%A Mauá, Denis Deratani
%A Finger, Marcelo
%Y Barbosa, Bryan Khelven da Silva
%Y Paes, Aline
%Y Felippo, Ariani Di
%S Proceedings of the 17th Brazilian Symposium in Information and Human Language Technology
%D 2026
%8 October
%I Association for Computational Linguistics
%C Cuiabá, Mato Grosso, Brazil
%F serras-etal-2026-compression
%X Compression-based linguistic complexity metrics enable cross-linguistic comparison without prior annotation. Their sensitivity to variation across languages and Portuguese registers highlights their applicability in NLP tasks. This study investigates their use as readability proxies and complementary features in Automatic Essay Scoring. We analyze how these metrics capture variation in essay quality across traits, genres, and educational levels in Brazilian Portuguese. In addition, we evaluate their sensitivity to differences between humanand AI-generated essays. Our results suggest that complexity metrics are effective (i) in differentiating educational levels, (ii) in detecting whether they were written by humans and (iii) as predictors of essay quality.
%R 10.5753/stil.2026.26572
%U https://aclanthology.org/2026.stil-1.32/
%U https://doi.org/10.5753/stil.2026.26572
%P 390-404
Markdown (Informal)
[Compression-Based Linguistic Complexity Metrics in Automatic Essay Scoring](https://aclanthology.org/2026.stil-1.32/) (Serras et al., STIL 2026)
ACL