@inproceedings{ruiz-fabo-etal-2026-automatic,
title = "Automatic Metrical Scansion of Poetry in a Low-Resource Setting",
author = "Ruiz Fabo, Pablo and
P{\'e}rez, Anxo Alonso and
Rodr{\'i}guez Fern{\'a}ndez, Pablo and
Gamallo, Pablo",
editor = "Montejo-Raez, Arturo and
Grisot, Cristina and
Blochowiak, Joanna and
Ljube{\v{s}}i{\'c}, Nikola and
Battaner, Elena and
Rigau, German",
booktitle = "Proceedings of Shaping Multilingual, Multimodal {AI} for the Social Sciences and Humanities ({LLM}s4{SSH}) @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma de Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.llms4ssh-1.12/",
doi = "10.63317/4pzd6u7388jm",
pages = "114--125",
abstract = "We present the first neural systems for automatic metrical scansion of poetry in Galician, a Romance language close to Portuguese and Spanish. The task is threefold: First, identifying metrical syllables based on lexical ones; both syllable series may differ given metrical licenses modifying a line{'}s syllable structure to enable stress-related rhythms. Second, identifying stress patterns, and third identifying the metrical syllable count, based on stressed positions. We manually annotated a corpus of 4,287 examples, a first in Galician, and fine-tuned an 8B-parameter LLM specialized in Galician and Portuguese, and two encoder{--}decoder models: ByT5, a token-free byte-to-byte model, and the multilingual mT5, which includes Galician. We also tested our recent symbolic scansion system. Several fine-tuning setups reached exact per-line accuracy above 90{\%} on our test-set at all three scansion subtasks, using orthographic syllables with explicit stress marks as input. Encoder{--}decoders performed better than the LLM. The token-free ByT5 was best, particularly when adding the two surrounding lines to the input. The symbolic system (89.9{\%} acc.) managed rare metaplasms infrequent in training data better than the neural ones, and the approaches can be seen as complementary."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="ruiz-fabo-etal-2026-automatic">
<titleInfo>
<title>Automatic Metrical Scansion of Poetry in a Low-Resource Setting</title>
</titleInfo>
<name type="personal">
<namePart type="given">Pablo</namePart>
<namePart type="family">Ruiz Fabo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anxo</namePart>
<namePart type="given">Alonso</namePart>
<namePart type="family">Pérez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pablo</namePart>
<namePart type="family">Rodríguez Fernández</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Pablo</namePart>
<namePart type="family">Gamallo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Shaping Multilingual, Multimodal AI for the Social Sciences and Humanities (LLMs4SSH) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Arturo</namePart>
<namePart type="family">Montejo-Raez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Cristina</namePart>
<namePart type="family">Grisot</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joanna</namePart>
<namePart type="family">Blochowiak</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nikola</namePart>
<namePart type="family">Ljubešić</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Battaner</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">German</namePart>
<namePart type="family">Rigau</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present the first neural systems for automatic metrical scansion of poetry in Galician, a Romance language close to Portuguese and Spanish. The task is threefold: First, identifying metrical syllables based on lexical ones; both syllable series may differ given metrical licenses modifying a line’s syllable structure to enable stress-related rhythms. Second, identifying stress patterns, and third identifying the metrical syllable count, based on stressed positions. We manually annotated a corpus of 4,287 examples, a first in Galician, and fine-tuned an 8B-parameter LLM specialized in Galician and Portuguese, and two encoder–decoder models: ByT5, a token-free byte-to-byte model, and the multilingual mT5, which includes Galician. We also tested our recent symbolic scansion system. Several fine-tuning setups reached exact per-line accuracy above 90% on our test-set at all three scansion subtasks, using orthographic syllables with explicit stress marks as input. Encoder–decoders performed better than the LLM. The token-free ByT5 was best, particularly when adding the two surrounding lines to the input. The symbolic system (89.9% acc.) managed rare metaplasms infrequent in training data better than the neural ones, and the approaches can be seen as complementary.</abstract>
<identifier type="citekey">ruiz-fabo-etal-2026-automatic</identifier>
<identifier type="doi">10.63317/4pzd6u7388jm</identifier>
<location>
<url>https://aclanthology.org/2026.llms4ssh-1.12/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>114</start>
<end>125</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Automatic Metrical Scansion of Poetry in a Low-Resource Setting
%A Ruiz Fabo, Pablo
%A Pérez, Anxo Alonso
%A Rodríguez Fernández, Pablo
%A Gamallo, Pablo
%Y Montejo-Raez, Arturo
%Y Grisot, Cristina
%Y Blochowiak, Joanna
%Y Ljubešić, Nikola
%Y Battaner, Elena
%Y Rigau, German
%S Proceedings of Shaping Multilingual, Multimodal AI for the Social Sciences and Humanities (LLMs4SSH) @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca (Spain)
%F ruiz-fabo-etal-2026-automatic
%X We present the first neural systems for automatic metrical scansion of poetry in Galician, a Romance language close to Portuguese and Spanish. The task is threefold: First, identifying metrical syllables based on lexical ones; both syllable series may differ given metrical licenses modifying a line’s syllable structure to enable stress-related rhythms. Second, identifying stress patterns, and third identifying the metrical syllable count, based on stressed positions. We manually annotated a corpus of 4,287 examples, a first in Galician, and fine-tuned an 8B-parameter LLM specialized in Galician and Portuguese, and two encoder–decoder models: ByT5, a token-free byte-to-byte model, and the multilingual mT5, which includes Galician. We also tested our recent symbolic scansion system. Several fine-tuning setups reached exact per-line accuracy above 90% on our test-set at all three scansion subtasks, using orthographic syllables with explicit stress marks as input. Encoder–decoders performed better than the LLM. The token-free ByT5 was best, particularly when adding the two surrounding lines to the input. The symbolic system (89.9% acc.) managed rare metaplasms infrequent in training data better than the neural ones, and the approaches can be seen as complementary.
%R 10.63317/4pzd6u7388jm
%U https://aclanthology.org/2026.llms4ssh-1.12/
%U https://doi.org/10.63317/4pzd6u7388jm
%P 114-125
Markdown (Informal)
[Automatic Metrical Scansion of Poetry in a Low-Resource Setting](https://aclanthology.org/2026.llms4ssh-1.12/) (Ruiz Fabo et al., LLMs4SSH 2026)
ACL
- Pablo Ruiz Fabo, Anxo Alonso Pérez, Pablo Rodríguez Fernández, and Pablo Gamallo. 2026. Automatic Metrical Scansion of Poetry in a Low-Resource Setting. In Proceedings of Shaping Multilingual, Multimodal AI for the Social Sciences and Humanities (LLMs4SSH) @ LREC 2026, pages 114–125, Palma de Mallorca (Spain). ELRA Language Resources Association (ELRA).