@inproceedings{arana-etal-2026-multimodal,
title = "Multimodal Large Language Models for Low-Resource Languages: A Case Study for {B}asque",
author = "Arana, Lukas and
Etxaniz, Julen and
Salaberria, Ander and
Azkune, Gorka",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.721/",
doi = "10.63317/2ry23e89ew5v",
pages = "9172--9187",
abstract = "Current Multimodal Large Language Models exhibit very strong performance for several demanding tasks. While commercial MLLMs deliver acceptable performance in low-resource languages, comparable results remain unattained within the open science community. In this paper, we aim to develop a strong MLLM for a low-resource language, namely Basque. For that purpose, we develop our own training and evaluation image-text datasets, leveraging state-of-the-art translation systems. Using two different Large Language Models as backbones, the Llama-3.1-Instruct model and a Basque-adapted variant called Latxa, we explore several data mixtures for training, encompassing Basque and English languages for both multimodal and text-only data. Evaluating our MLLMs for close-ended and open-ended generation tasks, we show that: i) low ratios of Basque multimodal data (around 20{\%}) are already enough to obtain solid results on Basque benchmarks, and ii) contrary to expected, a Basque instructed backbone LLM is not required to obtain a strong MLLM in Basque. Additionally, we specify the optimal data mixture strategy, the effects of multimodal data in text-only tasks, and analyze evaluation approaches for open-ended generation tasks. Our results pave the way to develop MLLMs for other low-resource languages by openly releasing our resources."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="arana-etal-2026-multimodal">
<titleInfo>
<title>Multimodal Large Language Models for Low-Resource Languages: A Case Study for Basque</title>
</titleInfo>
<name type="personal">
<namePart type="given">Lukas</namePart>
<namePart type="family">Arana</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Julen</namePart>
<namePart type="family">Etxaniz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ander</namePart>
<namePart type="family">Salaberria</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Gorka</namePart>
<namePart type="family">Azkune</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Current Multimodal Large Language Models exhibit very strong performance for several demanding tasks. While commercial MLLMs deliver acceptable performance in low-resource languages, comparable results remain unattained within the open science community. In this paper, we aim to develop a strong MLLM for a low-resource language, namely Basque. For that purpose, we develop our own training and evaluation image-text datasets, leveraging state-of-the-art translation systems. Using two different Large Language Models as backbones, the Llama-3.1-Instruct model and a Basque-adapted variant called Latxa, we explore several data mixtures for training, encompassing Basque and English languages for both multimodal and text-only data. Evaluating our MLLMs for close-ended and open-ended generation tasks, we show that: i) low ratios of Basque multimodal data (around 20%) are already enough to obtain solid results on Basque benchmarks, and ii) contrary to expected, a Basque instructed backbone LLM is not required to obtain a strong MLLM in Basque. Additionally, we specify the optimal data mixture strategy, the effects of multimodal data in text-only tasks, and analyze evaluation approaches for open-ended generation tasks. Our results pave the way to develop MLLMs for other low-resource languages by openly releasing our resources.</abstract>
<identifier type="citekey">arana-etal-2026-multimodal</identifier>
<identifier type="doi">10.63317/2ry23e89ew5v</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.721/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>9172</start>
<end>9187</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Multimodal Large Language Models for Low-Resource Languages: A Case Study for Basque
%A Arana, Lukas
%A Etxaniz, Julen
%A Salaberria, Ander
%A Azkune, Gorka
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F arana-etal-2026-multimodal
%X Current Multimodal Large Language Models exhibit very strong performance for several demanding tasks. While commercial MLLMs deliver acceptable performance in low-resource languages, comparable results remain unattained within the open science community. In this paper, we aim to develop a strong MLLM for a low-resource language, namely Basque. For that purpose, we develop our own training and evaluation image-text datasets, leveraging state-of-the-art translation systems. Using two different Large Language Models as backbones, the Llama-3.1-Instruct model and a Basque-adapted variant called Latxa, we explore several data mixtures for training, encompassing Basque and English languages for both multimodal and text-only data. Evaluating our MLLMs for close-ended and open-ended generation tasks, we show that: i) low ratios of Basque multimodal data (around 20%) are already enough to obtain solid results on Basque benchmarks, and ii) contrary to expected, a Basque instructed backbone LLM is not required to obtain a strong MLLM in Basque. Additionally, we specify the optimal data mixture strategy, the effects of multimodal data in text-only tasks, and analyze evaluation approaches for open-ended generation tasks. Our results pave the way to develop MLLMs for other low-resource languages by openly releasing our resources.
%R 10.63317/2ry23e89ew5v
%U https://aclanthology.org/2026.lrec-1.721/
%U https://doi.org/10.63317/2ry23e89ew5v
%P 9172-9187
Markdown (Informal)
[Multimodal Large Language Models for Low-Resource Languages: A Case Study for Basque](https://aclanthology.org/2026.lrec-1.721/) (Arana et al., LREC 2026)
ACL