@inproceedings{tadic-etal-2026-building,
title = "Building the v4 of the {C}roatian National Corpus",
author = "Tadi{\'c}, Marko and
{\v{S}}tefanec, Vanja and
Farka{\v{s}}, Da{\v{s}}a",
editor = "Ba{\'n}ski, Piotr and
Knight, Dawn and
Kupietz, Marc and
Witt, Andreas and
Wr{\'o}blewska, Alina",
booktitle = "Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cmlc-1.13/",
doi = "10.63317/48si6yrozisf",
pages = "80--83",
abstract = "It has been thirteen years since the release of the current version (v3) of the Croatian National Corpus (HNK). In terms of synchronicity in corpus linguistics, that many years may be considered quite some time. The preparatory phase for the composition of the new version of HNK (v4) has been going already for several years and in this paper we touch on several issues of concern. Apart of regular corpus parameters, e.g. text sources, text genres, coverage of language varieties, time span, we also discuss about metadata and linguistic annotation schemata. One of important technical prerequisites was the development of CorpRepo, a custom corpus data management system and file system, which enable us to do sustainable long-term maintenance of the data, and to produce newer versions of corpus more easily and more often. The selection of IPR-cleared data entails some restrictions and we give several examples of that kind of textual sources, but also discuss possible weaknesses of such approach to data selection. Regarding the linguistic annotation, the important shift is the decision to abandon the MulText East morphosyntacting descriptions and use solutions recommended by UD-initiative."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="tadic-etal-2026-building">
<titleInfo>
<title>Building the v4 of the Croatian National Corpus</title>
</titleInfo>
<name type="personal">
<namePart type="given">Marko</namePart>
<namePart type="family">Tadić</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Vanja</namePart>
<namePart type="family">Štefanec</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Daša</namePart>
<namePart type="family">Farkaš</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Piotr</namePart>
<namePart type="family">Bański</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dawn</namePart>
<namePart type="family">Knight</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marc</namePart>
<namePart type="family">Kupietz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andreas</namePart>
<namePart type="family">Witt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Wróblewska</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>It has been thirteen years since the release of the current version (v3) of the Croatian National Corpus (HNK). In terms of synchronicity in corpus linguistics, that many years may be considered quite some time. The preparatory phase for the composition of the new version of HNK (v4) has been going already for several years and in this paper we touch on several issues of concern. Apart of regular corpus parameters, e.g. text sources, text genres, coverage of language varieties, time span, we also discuss about metadata and linguistic annotation schemata. One of important technical prerequisites was the development of CorpRepo, a custom corpus data management system and file system, which enable us to do sustainable long-term maintenance of the data, and to produce newer versions of corpus more easily and more often. The selection of IPR-cleared data entails some restrictions and we give several examples of that kind of textual sources, but also discuss possible weaknesses of such approach to data selection. Regarding the linguistic annotation, the important shift is the decision to abandon the MulText East morphosyntacting descriptions and use solutions recommended by UD-initiative.</abstract>
<identifier type="citekey">tadic-etal-2026-building</identifier>
<identifier type="doi">10.63317/48si6yrozisf</identifier>
<location>
<url>https://aclanthology.org/2026.cmlc-1.13/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>80</start>
<end>83</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Building the v4 of the Croatian National Corpus
%A Tadić, Marko
%A Štefanec, Vanja
%A Farkaš, Daša
%Y Bański, Piotr
%Y Knight, Dawn
%Y Kupietz, Marc
%Y Witt, Andreas
%Y Wróblewska, Alina
%S Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F tadic-etal-2026-building
%X It has been thirteen years since the release of the current version (v3) of the Croatian National Corpus (HNK). In terms of synchronicity in corpus linguistics, that many years may be considered quite some time. The preparatory phase for the composition of the new version of HNK (v4) has been going already for several years and in this paper we touch on several issues of concern. Apart of regular corpus parameters, e.g. text sources, text genres, coverage of language varieties, time span, we also discuss about metadata and linguistic annotation schemata. One of important technical prerequisites was the development of CorpRepo, a custom corpus data management system and file system, which enable us to do sustainable long-term maintenance of the data, and to produce newer versions of corpus more easily and more often. The selection of IPR-cleared data entails some restrictions and we give several examples of that kind of textual sources, but also discuss possible weaknesses of such approach to data selection. Regarding the linguistic annotation, the important shift is the decision to abandon the MulText East morphosyntacting descriptions and use solutions recommended by UD-initiative.
%R 10.63317/48si6yrozisf
%U https://aclanthology.org/2026.cmlc-1.13/
%U https://doi.org/10.63317/48si6yrozisf
%P 80-83
Markdown (Informal)
[Building the v4 of the Croatian National Corpus](https://aclanthology.org/2026.cmlc-1.13/) (Tadić et al., CMLC 2026)
ACL
- Marko Tadić, Vanja Štefanec, and Daša Farkaš. 2026. Building the v4 of the Croatian National Corpus. In Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora, pages 80–83, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).