@inproceedings{almalki-etal-2026-obsidian,
title = "{OBSIDIAN}: An {OSINT}-Driven {NLP} Framework for Detecting Cyber-Physical Threats in {A}rabic Social Media",
author = "Almalki, Abdullah Saeed and
Chafik, Salmane and
Ezzini, Saad",
editor = "Mitkov, Ruslan and
Mu{\~n}oz, Rafael and
Lloret, Elena and
Ranasinghe, Tharindu and
Estevanell-Valladares, Ernesto L. and
Lamsiyah, Salima and
Montoyo, Andr{\'e}s and
Ezzini, Saad",
booktitle = "Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security",
month = jun,
year = "2026",
address = "Alicante, Spain",
publisher = "Department of Languages and Information Systems, University of Alicante",
url = "https://aclanthology.org/2026.nlpaics-1.9/",
pages = "88--97",
abstract = "Social networks have evolved into rich sources of Open-Source Intelligence (OSINT), enabling analysts to monitor unrestrained content expressing user activities, sentiments, and emerging behaviors. The immense use of these platforms has made it essential for cybersecurity and threat intelligence professionals to analyze and classify such content to proactively detect cyber-physical threats. While significant research has been conducted on the Arabic language regarding Hate Speech (HS) and Cyberbullying (CB), limited work has addressed Cyber Threat Intelligence (CTI) and OSINT-driven security classification in Arabic, despite their critical importance for early-warning systems and crisis response. In this paper, we introduce OBSIDIAN-AR, a novel real-world, large-scale dataset designed for detecting cyber-physical threats, comprising over 15,000 social media posts primarily from the Gulf region. The dataset is manually curated and annotated into five OSINT-relevant categories: Violence, Threat, Distress, Complaint, and Neutral. Using OBSIDIAN-AR, we fine-tune an Arabic BERT-based model named OBSIDIAN. This context-aware framework acts as a digital early-warning system, leveraging pretrained language representations and domain-specific knowledge derived from our high-quality dataset. Experimental results demonstrate that OBSIDIAN achieves strong performance, reaching up to 98{\%} accuracy on unseen data, proving its viability for modern cybersecurity operations."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="almalki-etal-2026-obsidian">
<titleInfo>
<title>OBSIDIAN: An OSINT-Driven NLP Framework for Detecting Cyber-Physical Threats in Arabic Social Media</title>
</titleInfo>
<name type="personal">
<namePart type="given">Abdullah</namePart>
<namePart type="given">Saeed</namePart>
<namePart type="family">Almalki</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Salmane</namePart>
<namePart type="family">Chafik</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-06</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security</title>
</titleInfo>
<name type="personal">
<namePart type="given">Ruslan</namePart>
<namePart type="family">Mitkov</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rafael</namePart>
<namePart type="family">Muñoz</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Elena</namePart>
<namePart type="family">Lloret</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Tharindu</namePart>
<namePart type="family">Ranasinghe</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ernesto</namePart>
<namePart type="given">L</namePart>
<namePart type="family">Estevanell-Valladares</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Salima</namePart>
<namePart type="family">Lamsiyah</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Andrés</namePart>
<namePart type="family">Montoyo</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Saad</namePart>
<namePart type="family">Ezzini</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>Department of Languages and Information Systems, University of Alicante</publisher>
<place>
<placeTerm type="text">Alicante, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Social networks have evolved into rich sources of Open-Source Intelligence (OSINT), enabling analysts to monitor unrestrained content expressing user activities, sentiments, and emerging behaviors. The immense use of these platforms has made it essential for cybersecurity and threat intelligence professionals to analyze and classify such content to proactively detect cyber-physical threats. While significant research has been conducted on the Arabic language regarding Hate Speech (HS) and Cyberbullying (CB), limited work has addressed Cyber Threat Intelligence (CTI) and OSINT-driven security classification in Arabic, despite their critical importance for early-warning systems and crisis response. In this paper, we introduce OBSIDIAN-AR, a novel real-world, large-scale dataset designed for detecting cyber-physical threats, comprising over 15,000 social media posts primarily from the Gulf region. The dataset is manually curated and annotated into five OSINT-relevant categories: Violence, Threat, Distress, Complaint, and Neutral. Using OBSIDIAN-AR, we fine-tune an Arabic BERT-based model named OBSIDIAN. This context-aware framework acts as a digital early-warning system, leveraging pretrained language representations and domain-specific knowledge derived from our high-quality dataset. Experimental results demonstrate that OBSIDIAN achieves strong performance, reaching up to 98% accuracy on unseen data, proving its viability for modern cybersecurity operations.</abstract>
<identifier type="citekey">almalki-etal-2026-obsidian</identifier>
<location>
<url>https://aclanthology.org/2026.nlpaics-1.9/</url>
</location>
<part>
<date>2026-06</date>
<extent unit="page">
<start>88</start>
<end>97</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T OBSIDIAN: An OSINT-Driven NLP Framework for Detecting Cyber-Physical Threats in Arabic Social Media
%A Almalki, Abdullah Saeed
%A Chafik, Salmane
%A Ezzini, Saad
%Y Mitkov, Ruslan
%Y Muñoz, Rafael
%Y Lloret, Elena
%Y Ranasinghe, Tharindu
%Y Estevanell-Valladares, Ernesto L.
%Y Lamsiyah, Salima
%Y Montoyo, Andrés
%Y Ezzini, Saad
%S Proceedings of the Second International Conference on Natural Language Processing and Artificial Intelligence for Cyber Security
%D 2026
%8 June
%I Department of Languages and Information Systems, University of Alicante
%C Alicante, Spain
%F almalki-etal-2026-obsidian
%X Social networks have evolved into rich sources of Open-Source Intelligence (OSINT), enabling analysts to monitor unrestrained content expressing user activities, sentiments, and emerging behaviors. The immense use of these platforms has made it essential for cybersecurity and threat intelligence professionals to analyze and classify such content to proactively detect cyber-physical threats. While significant research has been conducted on the Arabic language regarding Hate Speech (HS) and Cyberbullying (CB), limited work has addressed Cyber Threat Intelligence (CTI) and OSINT-driven security classification in Arabic, despite their critical importance for early-warning systems and crisis response. In this paper, we introduce OBSIDIAN-AR, a novel real-world, large-scale dataset designed for detecting cyber-physical threats, comprising over 15,000 social media posts primarily from the Gulf region. The dataset is manually curated and annotated into five OSINT-relevant categories: Violence, Threat, Distress, Complaint, and Neutral. Using OBSIDIAN-AR, we fine-tune an Arabic BERT-based model named OBSIDIAN. This context-aware framework acts as a digital early-warning system, leveraging pretrained language representations and domain-specific knowledge derived from our high-quality dataset. Experimental results demonstrate that OBSIDIAN achieves strong performance, reaching up to 98% accuracy on unseen data, proving its viability for modern cybersecurity operations.
%U https://aclanthology.org/2026.nlpaics-1.9/
%P 88-97
Markdown (Informal)
[OBSIDIAN: An OSINT-Driven NLP Framework for Detecting Cyber-Physical Threats in Arabic Social Media](https://aclanthology.org/2026.nlpaics-1.9/) (Almalki et al., NLPAICS 2026)
ACL