@inproceedings{castillo-sancho-etal-2026-evaluating,
title = "Evaluating Data Augmentation Strategies for Training {S}panish Misspelling Detection Models",
author = "Castillo-Sancho, Manuel and
Porta, Jordi and
G{\'o}mez-P{\'e}rez, Asunci{\'o}n",
editor = "Gorman, Kyle",
booktitle = "Proceedings of the Third Workshop on Computation and Written Language ({CAWL} 2026) @ {LREC} 2026",
month = jun,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cawl-1.7/",
doi = "10.63317/3mw3y4ovzwsz",
pages = "71--78",
abstract = "This paper evaluates three data augmentation strategies for training misspelling detection models in Spanish. Using the Spanish CORRSIC corpus of naturally occurring misspellings, we compare three misspelling generation methods: random perturbations, keyboard-based errors, and a statistical model derived from empirical edit patterns encoded as weighted finite-state transducers. We also analyze two word selection strategies (random and length-based) and two augmentation configurations designed to balance data diversity and reduce spurious correlations. This study shows that the statistical model produces misspellings most similar to real data, showing the lowest Jensen{--}Shannon divergence (0.148 nats) with the empirical distribution. In downstream detection experiments, performance improves with training size, and differences between word selection strategies remain minimal. Overall, the results highlight the value of statistically grounded misspelling generation for realistic and effective data augmentation in spell-checking tasks in Spanish."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="castillo-sancho-etal-2026-evaluating">
<titleInfo>
<title>Evaluating Data Augmentation Strategies for Training Spanish Misspelling Detection Models</title>
</titleInfo>
<name type="personal">
<namePart type="given">Manuel</namePart>
<namePart type="family">Castillo-Sancho</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jordi</namePart>
<namePart type="family">Porta</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Asunción</namePart>
<namePart type="family">Gómez-Pérez</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Third Workshop on Computation and Written Language (CAWL 2026) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Kyle</namePart>
<namePart type="family">Gorman</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper evaluates three data augmentation strategies for training misspelling detection models in Spanish. Using the Spanish CORRSIC corpus of naturally occurring misspellings, we compare three misspelling generation methods: random perturbations, keyboard-based errors, and a statistical model derived from empirical edit patterns encoded as weighted finite-state transducers. We also analyze two word selection strategies (random and length-based) and two augmentation configurations designed to balance data diversity and reduce spurious correlations. This study shows that the statistical model produces misspellings most similar to real data, showing the lowest Jensen–Shannon divergence (0.148 nats) with the empirical distribution. In downstream detection experiments, performance improves with training size, and differences between word selection strategies remain minimal. Overall, the results highlight the value of statistically grounded misspelling generation for realistic and effective data augmentation in spell-checking tasks in Spanish.</abstract>
<identifier type="citekey">castillo-sancho-etal-2026-evaluating</identifier>
<identifier type="doi">10.63317/3mw3y4ovzwsz</identifier>
<location>
<url>https://aclanthology.org/2026.cawl-1.7/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>71</start>
<end>78</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Evaluating Data Augmentation Strategies for Training Spanish Misspelling Detection Models
%A Castillo-Sancho, Manuel
%A Porta, Jordi
%A Gómez-Pérez, Asunción
%Y Gorman, Kyle
%S Proceedings of the Third Workshop on Computation and Written Language (CAWL 2026) @ LREC 2026
%D 2026
%8 June
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F castillo-sancho-etal-2026-evaluating
%X This paper evaluates three data augmentation strategies for training misspelling detection models in Spanish. Using the Spanish CORRSIC corpus of naturally occurring misspellings, we compare three misspelling generation methods: random perturbations, keyboard-based errors, and a statistical model derived from empirical edit patterns encoded as weighted finite-state transducers. We also analyze two word selection strategies (random and length-based) and two augmentation configurations designed to balance data diversity and reduce spurious correlations. This study shows that the statistical model produces misspellings most similar to real data, showing the lowest Jensen–Shannon divergence (0.148 nats) with the empirical distribution. In downstream detection experiments, performance improves with training size, and differences between word selection strategies remain minimal. Overall, the results highlight the value of statistically grounded misspelling generation for realistic and effective data augmentation in spell-checking tasks in Spanish.
%R 10.63317/3mw3y4ovzwsz
%U https://aclanthology.org/2026.cawl-1.7/
%U https://doi.org/10.63317/3mw3y4ovzwsz
%P 71-78
Markdown (Informal)
[Evaluating Data Augmentation Strategies for Training Spanish Misspelling Detection Models](https://aclanthology.org/2026.cawl-1.7/) (Castillo-Sancho et al., CAWL 2026)
ACL