@inproceedings{laaguidi-etal-2026-duo,
title = "{DUO}{\_}{DE} A1: An Annotated Corpus of Online Learning Material for Beginning Learners of {G}erman as a Foreign Language",
author = "La{\^a}guidi, Jammila and
Ruban, Vitaliia and
Laarmann-Quante, Ronja and
Drackert, Anastasia",
editor = {Barth, Florian and
Du, Keli and
Calvo Tello, Jos{\'e} and
Gen{\^e}t, Philippe and
Lendvai, Piroska and
Sch{\"o}ch, Christof and
Trippel, Thorsten},
booktitle = "Proceedings of Leveraging Derived Text Formats to Unlock Copyrighted Collections for Open Science ({DTF}) @ {LREC} 2026",
month = jun,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.dtf-1.7/",
doi = "10.63317/5mo2pqo4dkpa",
pages = "51--62",
abstract = "This paper describes the creation of DUO{\_}DE A1, a corpus based on A1-level learning material from the Deutsch-Uni Online (DUO) language courses for German as a foreign language. We split the material into small segments and manually annotated each with fine-grained information such as the type of segment (e.g. task description, description of grammar), the medium (e.g. text, table, audio), the text units it contains (e.g. words, phrases, sentences) and other special features (e.g. marking cloze texts). Furthermore, we automatically tokenized, POS tagged and lemmatized the corpus and compared the performance of three models on these steps for different kinds of segments. We publish the created corpus in a manner that respects copyright, releasing all structural features, metadata and POS tags."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="laaguidi-etal-2026-duo">
<titleInfo>
<title>DUO_DE A1: An Annotated Corpus of Online Learning Material for Beginning Learners of German as a Foreign Language</title>
</titleInfo>
<name type="personal">
<namePart type="given">Jammila</namePart>
<namePart type="family">Laâguidi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vitaliia</namePart>
<namePart type="family">Ruban</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ronja</namePart>
<namePart type="family">Laarmann-Quante</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Anastasia</namePart>
<namePart type="family">Drackert</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of Leveraging Derived Text Formats to Unlock Copyrighted Collections for Open Science (DTF) @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Florian</namePart>
<namePart type="family">Barth</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Keli</namePart>
<namePart type="family">Du</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">José</namePart>
<namePart type="family">Calvo Tello</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Philippe</namePart>
<namePart type="family">Genêt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Piroska</namePart>
<namePart type="family">Lendvai</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christof</namePart>
<namePart type="family">Schöch</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Thorsten</namePart>
<namePart type="family">Trippel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper describes the creation of DUO_DE A1, a corpus based on A1-level learning material from the Deutsch-Uni Online (DUO) language courses for German as a foreign language. We split the material into small segments and manually annotated each with fine-grained information such as the type of segment (e.g. task description, description of grammar), the medium (e.g. text, table, audio), the text units it contains (e.g. words, phrases, sentences) and other special features (e.g. marking cloze texts). Furthermore, we automatically tokenized, POS tagged and lemmatized the corpus and compared the performance of three models on these steps for different kinds of segments. We publish the created corpus in a manner that respects copyright, releasing all structural features, metadata and POS tags.</abstract>
<identifier type="citekey">laaguidi-etal-2026-duo</identifier>
<identifier type="doi">10.63317/5mo2pqo4dkpa</identifier>
<location>
<url>https://aclanthology.org/2026.dtf-1.7/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>51</start>
<end>62</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T DUO_DE A1: An Annotated Corpus of Online Learning Material for Beginning Learners of German as a Foreign Language
%A Laâguidi, Jammila
%A Ruban, Vitaliia
%A Laarmann-Quante, Ronja
%A Drackert, Anastasia
%Y Barth, Florian
%Y Du, Keli
%Y Calvo Tello, José
%Y Genêt, Philippe
%Y Lendvai, Piroska
%Y Schöch, Christof
%Y Trippel, Thorsten
%S Proceedings of Leveraging Derived Text Formats to Unlock Copyrighted Collections for Open Science (DTF) @ LREC 2026
%D 2026
%8 June
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F laaguidi-etal-2026-duo
%X This paper describes the creation of DUO_DE A1, a corpus based on A1-level learning material from the Deutsch-Uni Online (DUO) language courses for German as a foreign language. We split the material into small segments and manually annotated each with fine-grained information such as the type of segment (e.g. task description, description of grammar), the medium (e.g. text, table, audio), the text units it contains (e.g. words, phrases, sentences) and other special features (e.g. marking cloze texts). Furthermore, we automatically tokenized, POS tagged and lemmatized the corpus and compared the performance of three models on these steps for different kinds of segments. We publish the created corpus in a manner that respects copyright, releasing all structural features, metadata and POS tags.
%R 10.63317/5mo2pqo4dkpa
%U https://aclanthology.org/2026.dtf-1.7/
%U https://doi.org/10.63317/5mo2pqo4dkpa
%P 51-62
Markdown (Informal)
[DUO_DE A1: An Annotated Corpus of Online Learning Material for Beginning Learners of German as a Foreign Language](https://aclanthology.org/2026.dtf-1.7/) (Laâguidi et al., DTF 2026)
ACL