@inproceedings{alturki-2026-adabeval,
title = "{A}dab{E}val 2026 Task {B} : Multi-Label Classification of {A}rabic Politeness Criteria in Social Media Media",
author = "Alturki, Rand Abdullah",
editor = "Al-Khalifa, Hend and
El-Haj, Mo and
Ezzini, Saad",
booktitle = "The 7th Workshop on Open-Source {A}rabic Corpora and Processing Tools ({OSACT}7) with 5 Shared Tasks",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/2026.osact-1.22/",
doi = "10.63317/32uwkkvvdfq6",
pages = "185--190",
abstract = "We address the problem of multi-label classification of politeness and impoliteness criteria in Arabic social media posts, as defined in Subtask B of an Arabic politeness shared task (CITATION) The goal is to assign up to four labels from nine pragmatic categories, including Insult, Criticism, Respect, Prayers, and Hospitality, to each post. We first construct consistent multi-label annotations by mapping heterogeneous criterion strings into the official label set and analyzing their skewed distribution. To mitigate severe class imbalance, especially for rare categories such as Hospitality and Racism/Discrimination, we apply targeted oversampling of minority instances. Our modelling pipeline combines a TF{--}IDF + Logistic Regression baseline with two transformer-based encoders, MARBERT and AraBERT-twitter, trained for multi-label classification with Focal Loss. We then aggregate model outputs through a weighted ensemble and optimize per-class decision thresholds on a held-out validation set to improve macro-averaged F1. Experiments on the shared-task train/validation split show that the ensemble substantially outperforms the TF{--}IDF baseline and individual transformers, particularly on underrepresented categories, while maintaining competitive performance on frequent labels."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="alturki-2026-adabeval">
<titleInfo>
<title>AdabEval 2026 Task B : Multi-Label Classification of Arabic Politeness Criteria in Social Media Media</title>
</titleInfo>
<name type="personal">
<namePart type="given">Rand</namePart>
<namePart type="given">Abdullah</namePart>
<namePart type="family">Alturki</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks</title>
</titleInfo>
<name type="personal">
<namePart type="given">Hend</namePart>
<namePart type="family">Al-Khalifa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Association for Computational Linguistics</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>We address the problem of multi-label classification of politeness and impoliteness criteria in Arabic social media posts, as defined in Subtask B of an Arabic politeness shared task (CITATION) The goal is to assign up to four labels from nine pragmatic categories, including Insult, Criticism, Respect, Prayers, and Hospitality, to each post. We first construct consistent multi-label annotations by mapping heterogeneous criterion strings into the official label set and analyzing their skewed distribution. To mitigate severe class imbalance, especially for rare categories such as Hospitality and Racism/Discrimination, we apply targeted oversampling of minority instances. Our modelling pipeline combines a TF–IDF + Logistic Regression baseline with two transformer-based encoders, MARBERT and AraBERT-twitter, trained for multi-label classification with Focal Loss. We then aggregate model outputs through a weighted ensemble and optimize per-class decision thresholds on a held-out validation set to improve macro-averaged F1. Experiments on the shared-task train/validation split show that the ensemble substantially outperforms the TF–IDF baseline and individual transformers, particularly on underrepresented categories, while maintaining competitive performance on frequent labels.</abstract>
<identifier type="citekey">alturki-2026-adabeval</identifier>
<identifier type="doi">10.63317/32uwkkvvdfq6</identifier>
<location>
<url>https://aclanthology.org/2026.osact-1.22/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>185</start>
<end>190</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T AdabEval 2026 Task B : Multi-Label Classification of Arabic Politeness Criteria in Social Media Media
%A Alturki, Rand Abdullah
%Y Al-Khalifa, Hend
%Y El-Haj, Mo
%Y Ezzini, Saad
%S The 7th Workshop on Open-Source Arabic Corpora and Processing Tools (OSACT7) with 5 Shared Tasks
%D 2026
%8 May
%I Association for Computational Linguistics
%C Palma, Mallorca (Spain)
%F alturki-2026-adabeval
%X We address the problem of multi-label classification of politeness and impoliteness criteria in Arabic social media posts, as defined in Subtask B of an Arabic politeness shared task (CITATION) The goal is to assign up to four labels from nine pragmatic categories, including Insult, Criticism, Respect, Prayers, and Hospitality, to each post. We first construct consistent multi-label annotations by mapping heterogeneous criterion strings into the official label set and analyzing their skewed distribution. To mitigate severe class imbalance, especially for rare categories such as Hospitality and Racism/Discrimination, we apply targeted oversampling of minority instances. Our modelling pipeline combines a TF–IDF + Logistic Regression baseline with two transformer-based encoders, MARBERT and AraBERT-twitter, trained for multi-label classification with Focal Loss. We then aggregate model outputs through a weighted ensemble and optimize per-class decision thresholds on a held-out validation set to improve macro-averaged F1. Experiments on the shared-task train/validation split show that the ensemble substantially outperforms the TF–IDF baseline and individual transformers, particularly on underrepresented categories, while maintaining competitive performance on frequent labels.
%R 10.63317/32uwkkvvdfq6
%U https://aclanthology.org/2026.osact-1.22/
%U https://doi.org/10.63317/32uwkkvvdfq6
%P 185-190
Markdown (Informal)
[AdabEval 2026 Task B : Multi-Label Classification of Arabic Politeness Criteria in Social Media Media](https://aclanthology.org/2026.osact-1.22/) (Alturki, OSACT 2026)
ACL