@inproceedings{ibrahim-etal-2026-bigger,
title = "When Bigger Isn{'}t Better: Evaluating {LLM}s for {A}rabic Sentiment Analysis",
author = "Ibrahim, Mohamed and
Makki, Abdullah and
Barakat, Youssef and
Samy, Nour and
AlHumoud, Sarah",
editor = "Al-Khalifa, Hend and
El-Haj, Mo and
Ezzini, Saad",
booktitle = "The 7th Workshop on Open-Source {A}rabic Corpora and Processing Tools ({OSACT}7) with 5 Shared Tasks",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.osact-1.4/",
doi = "10.63317/2kimof4u6y8x",
pages = "35--39",
abstract = "This study evaluates the performance of a fine-tuned Arabic sentiment transformer (CAMeL-MSA) against eight large language models (LLMs). Using zero-shot prompting across six Arabic sentiment datasets, we compare a specialized, task-specific approach against generalized model capabilities. Results show that the fine-tuned baseline substantially outperformed all LLMs on five of the six datasets in both accuracy and Macro F1-score. While LLMs offer versatility, this comparison highlights the continued practical superiority of task-specific fine-tuning over zero-shot prompting."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="ibrahim-etal-2026-bigger">
<titleInfo>
<title>When Bigger Isn’t Better: Evaluating LLMs for Arabic Sentiment Analysis</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mohamed</namePart>
<namePart type="family">Ibrahim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Abdullah</namePart>
<namePart type="family">Makki</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Youssef</namePart>
<namePart type="family">Barakat</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nour</namePart>
<namePart type="family">Samy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Sarah</namePart>
<namePart type="family">AlHumoud</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hend</namePart>
<namePart type="family">Al-Khalifa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>This study evaluates the performance of a fine-tuned Arabic sentiment transformer (CAMeL-MSA) against eight large language models (LLMs). Using zero-shot prompting across six Arabic sentiment datasets, we compare a specialized, task-specific approach against generalized model capabilities. Results show that the fine-tuned baseline substantially outperformed all LLMs on five of the six datasets in both accuracy and Macro F1-score. While LLMs offer versatility, this comparison highlights the continued practical superiority of task-specific fine-tuning over zero-shot prompting.</abstract>
<identifier type="citekey">ibrahim-etal-2026-bigger</identifier>
<identifier type="doi">10.63317/2kimof4u6y8x</identifier>
<location>
<url>https://aclanthology.org/2026.osact-1.4/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>35</start>
<end>39</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T When Bigger Isn’t Better: Evaluating LLMs for Arabic Sentiment Analysis
%A Ibrahim, Mohamed
%A Makki, Abdullah
%A Barakat, Youssef
%A Samy, Nour
%A AlHumoud, Sarah
%Y Al-Khalifa, Hend
%Y El-Haj, Mo
%Y Ezzini, Saad
%S The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma, Mallorca (Spain)
%F ibrahim-etal-2026-bigger
%X This study evaluates the performance of a fine-tuned Arabic sentiment transformer (CAMeL-MSA) against eight large language models (LLMs). Using zero-shot prompting across six Arabic sentiment datasets, we compare a specialized, task-specific approach against generalized model capabilities. Results show that the fine-tuned baseline substantially outperformed all LLMs on five of the six datasets in both accuracy and Macro F1-score. While LLMs offer versatility, this comparison highlights the continued practical superiority of task-specific fine-tuning over zero-shot prompting.
%R 10.63317/2kimof4u6y8x
%U https://aclanthology.org/2026.osact-1.4/
%U https://doi.org/10.63317/2kimof4u6y8x
%P 35-39
Markdown (Informal)
[When Bigger Isn’t Better: Evaluating LLMs for Arabic Sentiment Analysis](https://aclanthology.org/2026.osact-1.4/) (Ibrahim et al., OSACT 2026)
ACL