@inproceedings{holdt-etal-2025-words,
title = "From Words to Action: A National Initiative to Overcome Data Scarcity for the {S}lovene {LLM}",
author = "Holdt, {\v{S}}pela Arhar and
Antloga, {\v{S}}pela and
Munda, Tina and
Pori, Eva and
Krek, Simon",
editor = "Holdt, {\v{S}}pela Arhar and
Ilinykh, Nikolai and
Scalvini, Barbara and
Bruton, Micaella and
Debess, Iben Nyholm and
Tudor, Crina Madalina",
booktitle = "Proceedings of the Third Workshop on Resources and Representations for Under-Resourced Languages and Domains (RESOURCEFUL-2025)",
month = mar,
year = "2025",
address = "Tallinn, Estonia",
publisher = "University of Tartu Library, Estonia",
url = "https://aclanthology.org/2025.resourceful-1.27/",
pages = "130--136",
ISBN = "978-9908-53-121-2",
abstract = "Large Language Models (LLMs) have demonstrated significant potential in natural language processing, but they depend on vast, diverse datasets, creating challenges for languages with limited resources. The paper presents a national initiative that addresses these challenges for Slovene. We outline strategies for large-scale text collection, including the creation of an online platform to engage the broader public in contributing texts and a communication campaign promoting openly accessible and transparently developed LLMs."
}
<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="holdt-etal-2025-words">
<titleInfo>
<title>From Words to Action: A National Initiative to Overcome Data Scarcity for the Slovene LLM</title>
</titleInfo>
<name type="personal">
<namePart type="given">Špela</namePart>
<namePart type="given">Arhar</namePart>
<namePart type="family">Holdt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Špela</namePart>
<namePart type="family">Antloga</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tina</namePart>
<namePart type="family">Munda</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Eva</namePart>
<namePart type="family">Pori</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2025-03</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Third Workshop on Resources and Representations for Under-Resourced Languages and Domains (RESOURCEFUL-2025)</title>
</titleInfo>
<name type="personal">
<namePart type="given">Špela</namePart>
<namePart type="given">Arhar</namePart>
<namePart type="family">Holdt</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nikolai</namePart>
<namePart type="family">Ilinykh</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Barbara</namePart>
<namePart type="family">Scalvini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Micaella</namePart>
<namePart type="family">Bruton</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Iben</namePart>
<namePart type="given">Nyholm</namePart>
<namePart type="family">Debess</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Crina</namePart>
<namePart type="given">Madalina</namePart>
<namePart type="family">Tudor</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>University of Tartu Library, Estonia</publisher>
<place>
<placeTerm type="text">Tallinn, Estonia</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
<identifier type="isbn">978-9908-53-121-2</identifier>
</relatedItem>
<abstract>Large Language Models (LLMs) have demonstrated significant potential in natural language processing, but they depend on vast, diverse datasets, creating challenges for languages with limited resources. The paper presents a national initiative that addresses these challenges for Slovene. We outline strategies for large-scale text collection, including the creation of an online platform to engage the broader public in contributing texts and a communication campaign promoting openly accessible and transparently developed LLMs.</abstract>
<identifier type="citekey">holdt-etal-2025-words</identifier>
<location>
<url>https://aclanthology.org/2025.resourceful-1.27/</url>
</location>
<part>
<date>2025-03</date>
<extent unit="page">
<start>130</start>
<end>136</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T From Words to Action: A National Initiative to Overcome Data Scarcity for the Slovene LLM
%A Holdt, Špela Arhar
%A Antloga, Špela
%A Munda, Tina
%A Pori, Eva
%A Krek, Simon
%Y Holdt, Špela Arhar
%Y Ilinykh, Nikolai
%Y Scalvini, Barbara
%Y Bruton, Micaella
%Y Debess, Iben Nyholm
%Y Tudor, Crina Madalina
%S Proceedings of the Third Workshop on Resources and Representations for Under-Resourced Languages and Domains (RESOURCEFUL-2025)
%D 2025
%8 March
%I University of Tartu Library, Estonia
%C Tallinn, Estonia
%@ 978-9908-53-121-2
%F holdt-etal-2025-words
%X Large Language Models (LLMs) have demonstrated significant potential in natural language processing, but they depend on vast, diverse datasets, creating challenges for languages with limited resources. The paper presents a national initiative that addresses these challenges for Slovene. We outline strategies for large-scale text collection, including the creation of an online platform to engage the broader public in contributing texts and a communication campaign promoting openly accessible and transparently developed LLMs.
%U https://aclanthology.org/2025.resourceful-1.27/
%P 130-136
Markdown (Informal)
[From Words to Action: A National Initiative to Overcome Data Scarcity for the Slovene LLM](https://aclanthology.org/2025.resourceful-1.27/) (Holdt et al., RESOURCEFUL 2025)
ACL