@inproceedings{hofenbitzer-etal-2026-german,
title = "The {G}erman Medical Text Corpus: Early 2026 Update",
author = {Hofenbitzer, Justin and
Lohr, Christina and
Meineke, Frank and
L{\"o}ffler, Markus and
Boeker, Martin},
editor = "Ba{\'n}ski, Piotr and
Knight, Dawn and
Kupietz, Marc and
Witt, Andreas and
Wr{\'o}blewska, Alina",
booktitle = "Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cmlc-1.16/",
doi = "10.63317/3xopdv4wdd93",
pages = "98--100",
abstract = "Clinical text resources are a central component for the study of medical language, as well as the training and evaluation of large language models, chatbots, and artificial intelligence systems supporting clinical routines. With the German Medical Text Corpus (GeMTeX), we are currently working on the largest shareable clinical document dataset in German. The multi-centric project ensures diversity across different university hospitals, clinical domains, and text sorts. After a thorough de-identification process, the clinical texts are semantically annotated using Snomed CT, a language-independent, standardized medical ontology. While the corpus is still under active development, it is accessible upon request under controlled access conditions. As of February 2026, GeMTeX comprises more than 15k documents and 20M tokens. We refer researchers interested in the resource to visit \url{https://kiinformatik.mri.tum.de/en/gemtex} or reach out to us via gemtex.mi@mh.tum.de."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hofenbitzer-etal-2026-german">
<titleInfo>
<title>The German Medical Text Corpus: Early 2026 Update</title>
</titleInfo>
<name type="personal">
<namePart type="given">Justin</namePart>
<namePart type="family">Hofenbitzer</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Christina</namePart>
<namePart type="family">Lohr</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Frank</namePart>
<namePart type="family">Meineke</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Markus</namePart>
<namePart type="family">Löffler</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Martin</namePart>
<namePart type="family">Boeker</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Piotr</namePart>
<namePart type="family">Bański</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dawn</namePart>
<namePart type="family">Knight</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marc</namePart>
<namePart type="family">Kupietz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andreas</namePart>
<namePart type="family">Witt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Wróblewska</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Clinical text resources are a central component for the study of medical language, as well as the training and evaluation of large language models, chatbots, and artificial intelligence systems supporting clinical routines. With the German Medical Text Corpus (GeMTeX), we are currently working on the largest shareable clinical document dataset in German. The multi-centric project ensures diversity across different university hospitals, clinical domains, and text sorts. After a thorough de-identification process, the clinical texts are semantically annotated using Snomed CT, a language-independent, standardized medical ontology. While the corpus is still under active development, it is accessible upon request under controlled access conditions. As of February 2026, GeMTeX comprises more than 15k documents and 20M tokens. We refer researchers interested in the resource to visit https://kiinformatik.mri.tum.de/en/gemtex or reach out to us via gemtex.mi@mh.tum.de.</abstract>
<identifier type="citekey">hofenbitzer-etal-2026-german</identifier>
<identifier type="doi">10.63317/3xopdv4wdd93</identifier>
<location>
<url>https://aclanthology.org/2026.cmlc-1.16/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>98</start>
<end>100</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T The German Medical Text Corpus: Early 2026 Update
%A Hofenbitzer, Justin
%A Lohr, Christina
%A Meineke, Frank
%A Löffler, Markus
%A Boeker, Martin
%Y Bański, Piotr
%Y Knight, Dawn
%Y Kupietz, Marc
%Y Witt, Andreas
%Y Wróblewska, Alina
%S Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F hofenbitzer-etal-2026-german
%X Clinical text resources are a central component for the study of medical language, as well as the training and evaluation of large language models, chatbots, and artificial intelligence systems supporting clinical routines. With the German Medical Text Corpus (GeMTeX), we are currently working on the largest shareable clinical document dataset in German. The multi-centric project ensures diversity across different university hospitals, clinical domains, and text sorts. After a thorough de-identification process, the clinical texts are semantically annotated using Snomed CT, a language-independent, standardized medical ontology. While the corpus is still under active development, it is accessible upon request under controlled access conditions. As of February 2026, GeMTeX comprises more than 15k documents and 20M tokens. We refer researchers interested in the resource to visit https://kiinformatik.mri.tum.de/en/gemtex or reach out to us via gemtex.mi@mh.tum.de.
%R 10.63317/3xopdv4wdd93
%U https://aclanthology.org/2026.cmlc-1.16/
%U https://doi.org/10.63317/3xopdv4wdd93
%P 98-100
Markdown (Informal)
[The German Medical Text Corpus: Early 2026 Update](https://aclanthology.org/2026.cmlc-1.16/) (Hofenbitzer et al., CMLC 2026)
ACL
- Justin Hofenbitzer, Christina Lohr, Frank Meineke, Markus Löffler, and Martin Boeker. 2026. The German Medical Text Corpus: Early 2026 Update. In Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora, pages 98–100, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).