@inproceedings{dahimi-etal-2026-foundation,
title = "How Foundation Models Behave for {A}rabic Image Captioning?",
author = "Dahimi, Khaoula and
Belabbaci, Amel and
Cherroun, Hadda and
Haouhat, Abdelhamid",
editor = "Al-Khalifa, Hend and
El-Haj, Mo and
Ezzini, Saad",
booktitle = "The 7th Workshop on Open-Source {A}rabic Corpora and Processing Tools ({OSACT}7) with 5 Shared Tasks",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.osact-1.6/",
doi = "10.63317/3bhwcpon3fv5",
pages = "49--58",
abstract = "Image captioning plays a crucial role in numerous applications, including educational systems. However, ensuring caption quality remains a significant challenge, particularly for morphologically rich, low-resource languages such as Arabic. We investigate an evaluation of Arabic image captioning using state-of-the-art multimodal foundation models. We systematically assess the performance of leading models{---}Gemini, Gemma, LLaMA, and Fanar. Our evaluation framework employs a diverse set of metrics spanning rule-based, learnable, visually-grounded, and LLM-based approaches to capture semantic accuracy, linguistic fluency, and hallucination detection. Experiments are conducted on two benchmark datasets: Flickr8k-Arabic and JEEM. Our findings reveal significant performance variations across models and evaluation metrics, highlighting the need for Arabic-specific optimization in multimodal architectures."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="dahimi-etal-2026-foundation">
<titleInfo>
<title>How Foundation Models Behave for Arabic Image Captioning?</title>
</titleInfo>
<name type="personal">
<namePart type="given">Khaoula</namePart>
<namePart type="family">Dahimi</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Amel</namePart>
<namePart type="family">Belabbaci</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Hadda</namePart>
<namePart type="family">Cherroun</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Abdelhamid</namePart>
<namePart type="family">Haouhat</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hend</namePart>
<namePart type="family">Al-Khalifa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Image captioning plays a crucial role in numerous applications, including educational systems. However, ensuring caption quality remains a significant challenge, particularly for morphologically rich, low-resource languages such as Arabic. We investigate an evaluation of Arabic image captioning using state-of-the-art multimodal foundation models. We systematically assess the performance of leading models—Gemini, Gemma, LLaMA, and Fanar. Our evaluation framework employs a diverse set of metrics spanning rule-based, learnable, visually-grounded, and LLM-based approaches to capture semantic accuracy, linguistic fluency, and hallucination detection. Experiments are conducted on two benchmark datasets: Flickr8k-Arabic and JEEM. Our findings reveal significant performance variations across models and evaluation metrics, highlighting the need for Arabic-specific optimization in multimodal architectures.</abstract>
<identifier type="citekey">dahimi-etal-2026-foundation</identifier>
<identifier type="doi">10.63317/3bhwcpon3fv5</identifier>
<location>
<url>https://aclanthology.org/2026.osact-1.6/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>49</start>
<end>58</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T How Foundation Models Behave for Arabic Image Captioning?
%A Dahimi, Khaoula
%A Belabbaci, Amel
%A Cherroun, Hadda
%A Haouhat, Abdelhamid
%Y Al-Khalifa, Hend
%Y El-Haj, Mo
%Y Ezzini, Saad
%S The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma, Mallorca (Spain)
%F dahimi-etal-2026-foundation
%X Image captioning plays a crucial role in numerous applications, including educational systems. However, ensuring caption quality remains a significant challenge, particularly for morphologically rich, low-resource languages such as Arabic. We investigate an evaluation of Arabic image captioning using state-of-the-art multimodal foundation models. We systematically assess the performance of leading models—Gemini, Gemma, LLaMA, and Fanar. Our evaluation framework employs a diverse set of metrics spanning rule-based, learnable, visually-grounded, and LLM-based approaches to capture semantic accuracy, linguistic fluency, and hallucination detection. Experiments are conducted on two benchmark datasets: Flickr8k-Arabic and JEEM. Our findings reveal significant performance variations across models and evaluation metrics, highlighting the need for Arabic-specific optimization in multimodal architectures.
%R 10.63317/3bhwcpon3fv5
%U https://aclanthology.org/2026.osact-1.6/
%U https://doi.org/10.63317/3bhwcpon3fv5
%P 49-58
Markdown (Informal)
[How Foundation Models Behave for Arabic Image Captioning?](https://aclanthology.org/2026.osact-1.6/) (Dahimi et al., OSACT 2026)
ACL
- Khaoula Dahimi, Amel Belabbaci, Hadda Cherroun, and Abdelhamid Haouhat. 2026. How Foundation Models Behave for Arabic Image Captioning?. In The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks, pages 49–58, Palma, Mallorca (Spain). Association for Computational Linguistics.