@inproceedings{kupietz-etal-2026-eureco,
title = "{E}u{R}e{C}o, {K}or{AP} and {D}e{R}e{K}o: Updates on Ingestion and Annotation Pipelines, Backend, Interfaces, Operation, and Corpora",
author = {Kupietz, Marc and
Diewald, Nils and
L{\"u}ngen, Harald and
Margaretha Illig, Eliza and
Stallkamp, Helge and
Tran, Uyen-Nhu and
Yaddehige, Rameela},
editor = "Ba{\'n}ski, Piotr and
Knight, Dawn and
Kupietz, Marc and
Witt, Andreas and
Wr{\'o}blewska, Alina",
booktitle = "Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cmlc-1.19/",
doi = "10.63317/2xcbb5knp2mi",
pages = "106--112",
abstract = "This paper reports on recent technical developments in the European Reference Corpus EuReCo and its current technical implementation based on the corpus search and analysis platform KorAP. We describe updates to the ingestion pipeline, including extensions to the TEI-to-KorAP-XML converter tei2korapxml and the KorAP tokenizer, as well as the newly introduced korapxmltool for annotation and index conversion. We further present Koral-Mapper, a service that enables cross-schema comparability of annotations and metadata at query time, and report on developments in the backend access control system Kustvakt, the web user interface Kalamar, API client libraries for R and Python that promote reproducibility and methodologically sound AI-assisted analysis, and containerized deployment. The corpora and languages currently represented in EuReCo are outlined, and the role of the German Reference Corpus DeReKo, including its metadata-driven virtual corpus design, predefined useful subcorpora, and TEI encoding, is discussed in detail. We further present the National Libraries as Corpus approach and DeLiKo-2025@DNB as its first full-scale proof of concept, and discuss the potential of this approach for extending EuReCo with comparable contemporary fiction corpora across European countries."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="kupietz-etal-2026-eureco">
<titleInfo>
<title>EuReCo, KorAP and DeReKo: Updates on Ingestion and Annotation Pipelines, Backend, Interfaces, Operation, and Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Marc</namePart>
<namePart type="family">Kupietz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nils</namePart>
<namePart type="family">Diewald</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Harald</namePart>
<namePart type="family">Lüngen</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eliza</namePart>
<namePart type="family">Margaretha Illig</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Helge</namePart>
<namePart type="family">Stallkamp</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Uyen-Nhu</namePart>
<namePart type="family">Tran</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rameela</namePart>
<namePart type="family">Yaddehige</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Piotr</namePart>
<namePart type="family">Bański</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dawn</namePart>
<namePart type="family">Knight</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marc</namePart>
<namePart type="family">Kupietz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andreas</namePart>
<namePart type="family">Witt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Wróblewska</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This paper reports on recent technical developments in the European Reference Corpus EuReCo and its current technical implementation based on the corpus search and analysis platform KorAP. We describe updates to the ingestion pipeline, including extensions to the TEI-to-KorAP-XML converter tei2korapxml and the KorAP tokenizer, as well as the newly introduced korapxmltool for annotation and index conversion. We further present Koral-Mapper, a service that enables cross-schema comparability of annotations and metadata at query time, and report on developments in the backend access control system Kustvakt, the web user interface Kalamar, API client libraries for R and Python that promote reproducibility and methodologically sound AI-assisted analysis, and containerized deployment. The corpora and languages currently represented in EuReCo are outlined, and the role of the German Reference Corpus DeReKo, including its metadata-driven virtual corpus design, predefined useful subcorpora, and TEI encoding, is discussed in detail. We further present the National Libraries as Corpus approach and DeLiKo-2025@DNB as its first full-scale proof of concept, and discuss the potential of this approach for extending EuReCo with comparable contemporary fiction corpora across European countries.</abstract>
<identifier type="citekey">kupietz-etal-2026-eureco</identifier>
<identifier type="doi">10.63317/2xcbb5knp2mi</identifier>
<location>
<url>https://aclanthology.org/2026.cmlc-1.19/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>106</start>
<end>112</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T EuReCo, KorAP and DeReKo: Updates on Ingestion and Annotation Pipelines, Backend, Interfaces, Operation, and Corpora
%A Kupietz, Marc
%A Diewald, Nils
%A Lüngen, Harald
%A Margaretha Illig, Eliza
%A Stallkamp, Helge
%A Tran, Uyen-Nhu
%A Yaddehige, Rameela
%Y Bański, Piotr
%Y Knight, Dawn
%Y Kupietz, Marc
%Y Witt, Andreas
%Y Wróblewska, Alina
%S Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F kupietz-etal-2026-eureco
%X This paper reports on recent technical developments in the European Reference Corpus EuReCo and its current technical implementation based on the corpus search and analysis platform KorAP. We describe updates to the ingestion pipeline, including extensions to the TEI-to-KorAP-XML converter tei2korapxml and the KorAP tokenizer, as well as the newly introduced korapxmltool for annotation and index conversion. We further present Koral-Mapper, a service that enables cross-schema comparability of annotations and metadata at query time, and report on developments in the backend access control system Kustvakt, the web user interface Kalamar, API client libraries for R and Python that promote reproducibility and methodologically sound AI-assisted analysis, and containerized deployment. The corpora and languages currently represented in EuReCo are outlined, and the role of the German Reference Corpus DeReKo, including its metadata-driven virtual corpus design, predefined useful subcorpora, and TEI encoding, is discussed in detail. We further present the National Libraries as Corpus approach and DeLiKo-2025@DNB as its first full-scale proof of concept, and discuss the potential of this approach for extending EuReCo with comparable contemporary fiction corpora across European countries.
%R 10.63317/2xcbb5knp2mi
%U https://aclanthology.org/2026.cmlc-1.19/
%U https://doi.org/10.63317/2xcbb5knp2mi
%P 106-112
Markdown (Informal)
[EuReCo, KorAP and DeReKo: Updates on Ingestion and Annotation Pipelines, Backend, Interfaces, Operation, and Corpora](https://aclanthology.org/2026.cmlc-1.19/) (Kupietz et al., CMLC 2026)
ACL
- Marc Kupietz, Nils Diewald, Harald Lüngen, Eliza Margaretha Illig, Helge Stallkamp, Uyen-Nhu Tran, and Rameela Yaddehige. 2026. EuReCo, KorAP and DeReKo: Updates on Ingestion and Annotation Pipelines, Backend, Interfaces, Operation, and Corpora. In Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora, pages 106–112, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).