@inproceedings{john-etal-2026-modeling,
title = "Modeling Word-Internal Structures: Morphological Segmentation Across 58 Languages",
author = "John, Vojt{\v{e}}ch and
{\v{Z}}abokrtsk{\'y}, Zden{\v{e}}k and
Reeves, Benjamin",
editor = "Hinrichs, Erhard and
Nivre, Joakim and
Osenova, Petya and
Pustejovsky, James and
Zinn, Claus",
booktitle = "Proceedings of the Workshop on Structured Linguistic Data and Evaluation ({SL}i{DE})",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.slide-1.16/",
doi = "10.63317/4qgccaokf8mz",
pages = "180--190",
abstract = "We present the largest multilingual experiment to date on word-to-morph segmentation, covering 58 typologically diverse languages. We describe a newly compiled collection of linguistically annotated resources for the task, providing broad coverage and enabling systematic cross-lingual evaluation. Second, we train two neural models on surface morphological segmentation, achieving 81{\%} average word accuracy on the original datasets, slightly outperforming previous methods. Experiments on custom test sets reveal substantial variation in performance, highlighting the need for further harmonization and more robust multilingual approaches."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="john-etal-2026-modeling">
<titleInfo>
<title>Modeling Word-Internal Structures: Morphological Segmentation Across 58 Languages</title>
</titleInfo>
<name type="personal">
<namePart type="given">Vojtěch</namePart>
<namePart type="family">John</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Zdeněk</namePart>
<namePart type="family">Žabokrtský</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Benjamin</namePart>
<namePart type="family">Reeves</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Workshop on Structured Linguistic Data and Evaluation (SLiDE)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Erhard</namePart>
<namePart type="family">Hinrichs</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Joakim</namePart>
<namePart type="family">Nivre</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Petya</namePart>
<namePart type="family">Osenova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">James</namePart>
<namePart type="family">Pustejovsky</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Claus</namePart>
<namePart type="family">Zinn</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present the largest multilingual experiment to date on word-to-morph segmentation, covering 58 typologically diverse languages. We describe a newly compiled collection of linguistically annotated resources for the task, providing broad coverage and enabling systematic cross-lingual evaluation. Second, we train two neural models on surface morphological segmentation, achieving 81% average word accuracy on the original datasets, slightly outperforming previous methods. Experiments on custom test sets reveal substantial variation in performance, highlighting the need for further harmonization and more robust multilingual approaches.</abstract>
<identifier type="citekey">john-etal-2026-modeling</identifier>
<identifier type="doi">10.63317/4qgccaokf8mz</identifier>
<location>
<url>https://aclanthology.org/2026.slide-1.16/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>180</start>
<end>190</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Modeling Word-Internal Structures: Morphological Segmentation Across 58 Languages
%A John, Vojtěch
%A Žabokrtský, Zdeněk
%A Reeves, Benjamin
%Y Hinrichs, Erhard
%Y Nivre, Joakim
%Y Osenova, Petya
%Y Pustejovsky, James
%Y Zinn, Claus
%S Proceedings of the Workshop on Structured Linguistic Data and Evaluation (SLiDE)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F john-etal-2026-modeling
%X We present the largest multilingual experiment to date on word-to-morph segmentation, covering 58 typologically diverse languages. We describe a newly compiled collection of linguistically annotated resources for the task, providing broad coverage and enabling systematic cross-lingual evaluation. Second, we train two neural models on surface morphological segmentation, achieving 81% average word accuracy on the original datasets, slightly outperforming previous methods. Experiments on custom test sets reveal substantial variation in performance, highlighting the need for further harmonization and more robust multilingual approaches.
%R 10.63317/4qgccaokf8mz
%U https://aclanthology.org/2026.slide-1.16/
%U https://doi.org/10.63317/4qgccaokf8mz
%P 180-190
Markdown (Informal)
[Modeling Word-Internal Structures: Morphological Segmentation Across 58 Languages](https://aclanthology.org/2026.slide-1.16/) (John et al., SLiDE 2026)
ACL