@inproceedings{koeva-stoyanova-2026-recent,
title = "Recent Developments of the {B}ulgarian National Corpus",
author = "Koeva, Svetla Peneva and
Stoyanova, Ivelina",
editor = "Ba{\'n}ski, Piotr and
Knight, Dawn and
Kupietz, Marc and
Witt, Andreas and
Wr{\'o}blewska, Alina",
booktitle = "Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.cmlc-1.10/",
doi = "10.63317/3m95ohtw7mjs",
pages = "71--75",
abstract = "We present recent developments in the Bulgarian National Corpus, including data collection from various sources, cleaning of diverse datasets, enrichment with multimodal data, and extensive metadata, which resulted in the development of IfGPT, a large BulNC-based dataset. Typical methods for distributing the BulNC-based dataset are briefly described, with emphasis on effective searching within the metadata stored in a graph database."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="koeva-stoyanova-2026-recent">
<titleInfo>
<title>Recent Developments of the Bulgarian National Corpus</title>
</titleInfo>
<name type="personal">
<namePart type="given">Svetla</namePart>
<namePart type="given">Peneva</namePart>
<namePart type="family">Koeva</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ivelina</namePart>
<namePart type="family">Stoyanova</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora</title>
</titleInfo>
<name type="personal">
<namePart type="given">Piotr</namePart>
<namePart type="family">Bański</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Dawn</namePart>
<namePart type="family">Knight</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Marc</namePart>
<namePart type="family">Kupietz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andreas</namePart>
<namePart type="family">Witt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Alina</namePart>
<namePart type="family">Wróblewska</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We present recent developments in the Bulgarian National Corpus, including data collection from various sources, cleaning of diverse datasets, enrichment with multimodal data, and extensive metadata, which resulted in the development of IfGPT, a large BulNC-based dataset. Typical methods for distributing the BulNC-based dataset are briefly described, with emphasis on effective searching within the metadata stored in a graph database.</abstract>
<identifier type="citekey">koeva-stoyanova-2026-recent</identifier>
<identifier type="doi">10.63317/3m95ohtw7mjs</identifier>
<location>
<url>https://aclanthology.org/2026.cmlc-1.10/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>71</start>
<end>75</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Recent Developments of the Bulgarian National Corpus
%A Koeva, Svetla Peneva
%A Stoyanova, Ivelina
%Y Bański, Piotr
%Y Knight, Dawn
%Y Kupietz, Marc
%Y Witt, Andreas
%Y Wróblewska, Alina
%S Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F koeva-stoyanova-2026-recent
%X We present recent developments in the Bulgarian National Corpus, including data collection from various sources, cleaning of diverse datasets, enrichment with multimodal data, and extensive metadata, which resulted in the development of IfGPT, a large BulNC-based dataset. Typical methods for distributing the BulNC-based dataset are briefly described, with emphasis on effective searching within the metadata stored in a graph database.
%R 10.63317/3m95ohtw7mjs
%U https://aclanthology.org/2026.cmlc-1.10/
%U https://doi.org/10.63317/3m95ohtw7mjs
%P 71-75
Markdown (Informal)
[Recent Developments of the Bulgarian National Corpus](https://aclanthology.org/2026.cmlc-1.10/) (Koeva & Stoyanova, CMLC 2026)
ACL
- Svetla Peneva Koeva and Ivelina Stoyanova. 2026. Recent Developments of the Bulgarian National Corpus. In Proceedings of the 12th Workshop on Challenges in the Management of Large Corpora, pages 71–75, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).