@inproceedings{stemle-2026-treebank,
title = "From Treebank Metadata to Sentence-Level Genre in {U}niversal {D}ependencies: A Reproducible, Versioned Resource",
author = "Stemle, Egon",
editor = {{\c{C}}{\"o}ltekin, {\c{C}}a{\u{g}}r{\i} and
Dobrovoljc, Kaja},
booktitle = "Proceedings of the Ninth Workshop on {U}niversal {D}ependencies ({UDW} 2026)",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.udw-1.24/",
doi = "10.63317/22ivc8whfgue",
pages = "268--276",
abstract = "We release a sentence-level genre layer for Universal Dependencies as a separate, joinable dataset, computed across UD revisions and linked back to the underlying treebanks via a release-aware composite key comprising treebank, split, sent{\_}id, and UD release metadata. The annotations are derived rather than authoritative and are accompanied by provenance and uncertainty indicators, enabling downstream users to choose appropriate precision-coverage trade-offs and to re-run the pipeline as UD evolves. To support both parity tracking and deployment-oriented interpretation, we report results under two complementary regimes: a fixed-partition setting aligned with earlier protocols, and a language-grouped 10-fold generalisation setting that highlights cross-language heterogeneity and anchor sparsity as operational constraints. The resulting resource is intended to make genre a practical control variable for UD-based experimentation, including genre-stratified evaluation and training data selection for POS tagging and parsing, where performance varies substantially across text types. Finally, we note that reduced genre spaces aligned with recurring robustness profiles (e.g. transcribed speech versus interactional web/social text versus edited prose/news) appear pragmatically useful, but should be treated as a community coordination task implemented through explicit, versioned mapping tables."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="stemle-2026-treebank">
<titleInfo>
<title>From Treebank Metadata to Sentence-Level Genre in Universal Dependencies: A Reproducible, Versioned Resource</title>
</titleInfo>
<name type="personal">
<namePart type="given">Egon</namePart>
<namePart type="family">Stemle</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Ninth Workshop on Universal Dependencies (UDW 2026)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Çağrı</namePart>
<namePart type="family">Çöltekin</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Kaja</namePart>
<namePart type="family">Dobrovoljc</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We release a sentence-level genre layer for Universal Dependencies as a separate, joinable dataset, computed across UD revisions and linked back to the underlying treebanks via a release-aware composite key comprising treebank, split, sent_id, and UD release metadata. The annotations are derived rather than authoritative and are accompanied by provenance and uncertainty indicators, enabling downstream users to choose appropriate precision-coverage trade-offs and to re-run the pipeline as UD evolves. To support both parity tracking and deployment-oriented interpretation, we report results under two complementary regimes: a fixed-partition setting aligned with earlier protocols, and a language-grouped 10-fold generalisation setting that highlights cross-language heterogeneity and anchor sparsity as operational constraints. The resulting resource is intended to make genre a practical control variable for UD-based experimentation, including genre-stratified evaluation and training data selection for POS tagging and parsing, where performance varies substantially across text types. Finally, we note that reduced genre spaces aligned with recurring robustness profiles (e.g. transcribed speech versus interactional web/social text versus edited prose/news) appear pragmatically useful, but should be treated as a community coordination task implemented through explicit, versioned mapping tables.</abstract>
<identifier type="citekey">stemle-2026-treebank</identifier>
<identifier type="doi">10.63317/22ivc8whfgue</identifier>
<location>
<url>https://aclanthology.org/2026.udw-1.24/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>268</start>
<end>276</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T From Treebank Metadata to Sentence-Level Genre in Universal Dependencies: A Reproducible, Versioned Resource
%A Stemle, Egon
%Y Çöltekin, Çağrı
%Y Dobrovoljc, Kaja
%S Proceedings of the Ninth Workshop on Universal Dependencies (UDW 2026)
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma de Mallorca, Spain
%F stemle-2026-treebank
%X We release a sentence-level genre layer for Universal Dependencies as a separate, joinable dataset, computed across UD revisions and linked back to the underlying treebanks via a release-aware composite key comprising treebank, split, sent_id, and UD release metadata. The annotations are derived rather than authoritative and are accompanied by provenance and uncertainty indicators, enabling downstream users to choose appropriate precision-coverage trade-offs and to re-run the pipeline as UD evolves. To support both parity tracking and deployment-oriented interpretation, we report results under two complementary regimes: a fixed-partition setting aligned with earlier protocols, and a language-grouped 10-fold generalisation setting that highlights cross-language heterogeneity and anchor sparsity as operational constraints. The resulting resource is intended to make genre a practical control variable for UD-based experimentation, including genre-stratified evaluation and training data selection for POS tagging and parsing, where performance varies substantially across text types. Finally, we note that reduced genre spaces aligned with recurring robustness profiles (e.g. transcribed speech versus interactional web/social text versus edited prose/news) appear pragmatically useful, but should be treated as a community coordination task implemented through explicit, versioned mapping tables.
%R 10.63317/22ivc8whfgue
%U https://aclanthology.org/2026.udw-1.24/
%U https://doi.org/10.63317/22ivc8whfgue
%P 268-276
Markdown (Informal)
[From Treebank Metadata to Sentence-Level Genre in Universal Dependencies: A Reproducible, Versioned Resource](https://aclanthology.org/2026.udw-1.24/) (Stemle, UDW 2026)
ACL