@inproceedings{hassan-etal-2026-digilians,
title = "Digilians at {N}akba{V}irality Shared Task: Bidirectional Cross-Attention for Multimodal Virality Prediction",
author = "Hassan, Ahmed Eid and
Mohamed, Noureldeen H. and
Fawzy, Abdelrhman M. and
Abdelghany, Mohamed A. and
Hassan, Ahmed A. and
Qassim, Ahmed S. and
Abd El Sayed, Fady A. and
Mohamed, Mohamed H. and
Mohamed, Rahma M. and
Abou-Attia, Arwa M. and
Mohamed, Mayar M. and
Sawla, Shahd A. and
Mahmoud, Shahd O.",
editor = "Jarrar, Mustafa and
El-Haj, Mo and
Haddad, Amal and
Atiani, Serin and
Abudalfa, Shadi and
Regier, Terry and
Rayson, Paul and
Sima{'}an, Khalil and
Mansour, Camille",
booktitle = "Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ {LREC} 2026",
month = may,
year = "2026",
address = "Palma, Mallorca (Spain)",
publisher = "ELRA Language Resources Association (ELRA)",
url = "https://aclanthology.org/2026.nakbanlp-1.41/",
doi = "10.63317/3tt97yspawfn",
pages = "269--272",
abstract = "The NakbaVirality shared task focuses on multimodal virality prediction using a dataset of 2,600 multilingual posts collected from X and Reddit. In this work, we propose a multimodal architecture that combines XLM-RoBERTa for text encoding and a Vision Transformer (ViT) for image representation. The extracted features are aligned through bidirectional cross-attention to capture interactions between textual and visual modalities. To address the class imbalance present in the dataset, we apply focal loss, class weighting, and targeted data augmentation for the minority class. Additionally, layer-wise learning rate scheduling is used to stabilize fine-tuning of the pretrained encoders. Experimental results show that the proposed system achieves an accuracy of 0.6009 on the hidden test set, ranking 4th among 29 participating teams (107 total submissions). These results highlight the effectiveness of cross-modal attention mechanisms for modeling multimodal signals in high-stakes discourse."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hassan-etal-2026-digilians">
<titleInfo>
<title>Digilians at NakbaVirality Shared Task: Bidirectional Cross-Attention for Multimodal Virality Prediction</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ahmed</namePart>
<namePart type="given">Eid</namePart>
<namePart type="family">Hassan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Noureldeen</namePart>
<namePart type="given">H</namePart>
<namePart type="family">Mohamed</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Abdelrhman</namePart>
<namePart type="given">M</namePart>
<namePart type="family">Fawzy</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mohamed</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Abdelghany</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ahmed</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Hassan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ahmed</namePart>
<namePart type="given">S</namePart>
<namePart type="family">Qassim</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Fady</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Abd El Sayed</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mohamed</namePart>
<namePart type="given">H</namePart>
<namePart type="family">Mohamed</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rahma</namePart>
<namePart type="given">M</namePart>
<namePart type="family">Mohamed</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Arwa</namePart>
<namePart type="given">M</namePart>
<namePart type="family">Abou-Attia</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mayar</namePart>
<namePart type="given">M</namePart>
<namePart type="family">Mohamed</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shahd</namePart>
<namePart type="given">A</namePart>
<namePart type="family">Sawla</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shahd</namePart>
<namePart type="given">O</namePart>
<namePart type="family">Mahmoud</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ LREC 2026</title>
</titleInfo>
<name type="personal">
<namePart type="given">Mustafa</namePart>
<namePart type="family">Jarrar</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mo</namePart>
<namePart type="family">El-Haj</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Amal</namePart>
<namePart type="family">Haddad</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Serin</namePart>
<namePart type="family">Atiani</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Shadi</namePart>
<namePart type="family">Abudalfa</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Terry</namePart>
<namePart type="family">Regier</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Paul</namePart>
<namePart type="family">Rayson</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Khalil</namePart>
<namePart type="family">Sima’an</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Camille</namePart>
<namePart type="family">Mansour</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resources Association (ELRA)</publisher>
<place>
<placeTerm type="text">Palma, Mallorca (Spain)</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>The NakbaVirality shared task focuses on multimodal virality prediction using a dataset of 2,600 multilingual posts collected from X and Reddit. In this work, we propose a multimodal architecture that combines XLM-RoBERTa for text encoding and a Vision Transformer (ViT) for image representation. The extracted features are aligned through bidirectional cross-attention to capture interactions between textual and visual modalities. To address the class imbalance present in the dataset, we apply focal loss, class weighting, and targeted data augmentation for the minority class. Additionally, layer-wise learning rate scheduling is used to stabilize fine-tuning of the pretrained encoders. Experimental results show that the proposed system achieves an accuracy of 0.6009 on the hidden test set, ranking 4th among 29 participating teams (107 total submissions). These results highlight the effectiveness of cross-modal attention mechanisms for modeling multimodal signals in high-stakes discourse.</abstract>
<identifier type="citekey">hassan-etal-2026-digilians</identifier>
<identifier type="doi">10.63317/3tt97yspawfn</identifier>
<location>
<url>https://aclanthology.org/2026.nakbanlp-1.41/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>269</start>
<end>272</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Digilians at NakbaVirality Shared Task: Bidirectional Cross-Attention for Multimodal Virality Prediction
%A Hassan, Ahmed Eid
%A Mohamed, Noureldeen H.
%A Fawzy, Abdelrhman M.
%A Abdelghany, Mohamed A.
%A Hassan, Ahmed A.
%A Qassim, Ahmed S.
%A Abd El Sayed, Fady A.
%A Mohamed, Mohamed H.
%A Mohamed, Rahma M.
%A Abou-Attia, Arwa M.
%A Mohamed, Mayar M.
%A Sawla, Shahd A.
%A Mahmoud, Shahd O.
%Y Jarrar, Mustafa
%Y El-Haj, Mo
%Y Haddad, Amal
%Y Atiani, Serin
%Y Abudalfa, Shadi
%Y Regier, Terry
%Y Rayson, Paul
%Y Sima’an, Khalil
%Y Mansour, Camille
%S Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ LREC 2026
%D 2026
%8 May
%I ELRA Language Resources Association (ELRA)
%C Palma, Mallorca (Spain)
%F hassan-etal-2026-digilians
%X The NakbaVirality shared task focuses on multimodal virality prediction using a dataset of 2,600 multilingual posts collected from X and Reddit. In this work, we propose a multimodal architecture that combines XLM-RoBERTa for text encoding and a Vision Transformer (ViT) for image representation. The extracted features are aligned through bidirectional cross-attention to capture interactions between textual and visual modalities. To address the class imbalance present in the dataset, we apply focal loss, class weighting, and targeted data augmentation for the minority class. Additionally, layer-wise learning rate scheduling is used to stabilize fine-tuning of the pretrained encoders. Experimental results show that the proposed system achieves an accuracy of 0.6009 on the hidden test set, ranking 4th among 29 participating teams (107 total submissions). These results highlight the effectiveness of cross-modal attention mechanisms for modeling multimodal signals in high-stakes discourse.
%R 10.63317/3tt97yspawfn
%U https://aclanthology.org/2026.nakbanlp-1.41/
%U https://doi.org/10.63317/3tt97yspawfn
%P 269-272
Markdown (Informal)
[Digilians at NakbaVirality Shared Task: Bidirectional Cross-Attention for Multimodal Virality Prediction](https://aclanthology.org/2026.nakbanlp-1.41/) (Hassan et al., NakbaNLP 2026)
ACL
- Ahmed Eid Hassan, Noureldeen H. Mohamed, Abdelrhman M. Fawzy, Mohamed A. Abdelghany, Ahmed A. Hassan, Ahmed S. Qassim, Fady A. Abd El Sayed, Mohamed H. Mohamed, Rahma M. Mohamed, Arwa M. Abou-Attia, Mayar M. Mohamed, Shahd A. Sawla, and Shahd O. Mahmoud. 2026. Digilians at NakbaVirality Shared Task: Bidirectional Cross-Attention for Multimodal Virality Prediction. In Proceedings of the 2nd International Workshop on Nakba Narratives as Language Resources @ LREC 2026, pages 269–272, Palma, Mallorca (Spain). ELRA Language Resources Association (ELRA).