@inproceedings{alshahrani-etal-2026-linguarabic,
title = "{L}ingu{A}rabic at {A}ra{S}ent{E}val 2026: {MARBERT} for Multi-Dialect {A}rabic Sentiment Analysis",
author = "Alshahrani, Norah Saud and
Al-Qarni, Elham Abdullah and
Alshomrani, Shatha Hussan",
editor = "Al-Khalifa, Hend and
El-Haj, Mo and
Ezzini, Saad",
booktitle = "The 7th Workshop on Open-Source {A}rabic Corpora and Processing Tools ({OSACT}7) with 5 Shared Tasks",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.osact-1.43/",
doi = "10.63317/2gbi3ds9jzqm",
pages = "302--305",
abstract = "Sentiment analysis for Arabic dialects remains challenging due to substantial linguistic variation across dialects and the expansion of informal language in user-generated content. The AraSentEval 2026 shared task introduces a multi-dialect benchmark designed to evaluate sentiment classification systems on real-world Arabic data. In this paper, we present LinguArabic{'}s submission to the sentiment classification track of AraSentEval 2026. Our approach is based on fine-tuning MARBERT, a transformer model pre-trained on large-scale Arabic social media data that captures diverse dialectal patterns. To improve model robustness, we incorporate a multi-stage preprocessing pipeline that includes text normalization, dialect-aware lexical mapping, and confidence-based prediction adjustment. We specifically investigate the impact of advanced normalization rules in reducing lexical sparsity across various regional dialects. Experimental results show that the proposed system achieves a Macro F1-score of 0.8333 on the offcial evaluation set. Our findings highlight the importance of dialect-aware pretraining and preprocessing strategies for improving sentiment classification performance across diverse Arabic dialects, providing a scalable framework for real-world Arabic NLP applications."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="alshahrani-etal-2026-linguarabic">
<titleInfo>
<title>LinguArabic at AraSentEval 2026: MARBERT for Multi-Dialect Arabic Sentiment Analysis</title>
</titleInfo>
<name type="personal">
<namePart type="given">Norah</namePart>
<namePart type="given">Saud</namePart>
<namePart type="family">Alshahrani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elham</namePart>
<namePart type="given">Abdullah</namePart>
<namePart type="family">Al-Qarni</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shatha</namePart>
<namePart type="given">Hussan</namePart>
<namePart type="family">Alshomrani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hend</namePart>
<namePart type="family">Al-Khalifa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Sentiment analysis for Arabic dialects remains challenging due to substantial linguistic variation across dialects and the expansion of informal language in user-generated content. The AraSentEval 2026 shared task introduces a multi-dialect benchmark designed to evaluate sentiment classification systems on real-world Arabic data. In this paper, we present LinguArabic’s submission to the sentiment classification track of AraSentEval 2026. Our approach is based on fine-tuning MARBERT, a transformer model pre-trained on large-scale Arabic social media data that captures diverse dialectal patterns. To improve model robustness, we incorporate a multi-stage preprocessing pipeline that includes text normalization, dialect-aware lexical mapping, and confidence-based prediction adjustment. We specifically investigate the impact of advanced normalization rules in reducing lexical sparsity across various regional dialects. Experimental results show that the proposed system achieves a Macro F1-score of 0.8333 on the offcial evaluation set. Our findings highlight the importance of dialect-aware pretraining and preprocessing strategies for improving sentiment classification performance across diverse Arabic dialects, providing a scalable framework for real-world Arabic NLP applications.</abstract>
<identifier type="citekey">alshahrani-etal-2026-linguarabic</identifier>
<identifier type="doi">10.63317/2gbi3ds9jzqm</identifier>
<location>
<url>https://aclanthology.org/2026.osact-1.43/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>302</start>
<end>305</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T LinguArabic at AraSentEval 2026: MARBERT for Multi-Dialect Arabic Sentiment Analysis
%A Alshahrani, Norah Saud
%A Al-Qarni, Elham Abdullah
%A Alshomrani, Shatha Hussan
%Y Al-Khalifa, Hend
%Y El-Haj, Mo
%Y Ezzini, Saad
%S The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma, Mallorca (Spain)
%F alshahrani-etal-2026-linguarabic
%X Sentiment analysis for Arabic dialects remains challenging due to substantial linguistic variation across dialects and the expansion of informal language in user-generated content. The AraSentEval 2026 shared task introduces a multi-dialect benchmark designed to evaluate sentiment classification systems on real-world Arabic data. In this paper, we present LinguArabic’s submission to the sentiment classification track of AraSentEval 2026. Our approach is based on fine-tuning MARBERT, a transformer model pre-trained on large-scale Arabic social media data that captures diverse dialectal patterns. To improve model robustness, we incorporate a multi-stage preprocessing pipeline that includes text normalization, dialect-aware lexical mapping, and confidence-based prediction adjustment. We specifically investigate the impact of advanced normalization rules in reducing lexical sparsity across various regional dialects. Experimental results show that the proposed system achieves a Macro F1-score of 0.8333 on the offcial evaluation set. Our findings highlight the importance of dialect-aware pretraining and preprocessing strategies for improving sentiment classification performance across diverse Arabic dialects, providing a scalable framework for real-world Arabic NLP applications.
%R 10.63317/2gbi3ds9jzqm
%U https://aclanthology.org/2026.osact-1.43/
%U https://doi.org/10.63317/2gbi3ds9jzqm
%P 302-305
Markdown (Informal)
[LinguArabic at AraSentEval 2026: MARBERT for Multi-Dialect Arabic Sentiment Analysis](https://aclanthology.org/2026.osact-1.43/) (Alshahrani et al., OSACT 2026)
ACL