@inproceedings{hu-etal-2026-harnessing,
title = "Harnessing Synergy in Context and Emoji for Joint Detection of Harmful Online Content in Multi-turn Conversations",
author = "Hu, Feiyan and
Byrne, Ciara Anne and
Zhou, Jiang and
Maycock, Rena and
Langan, Mark",
editor = "Piperidis, Stelios and
Bel, N{\'u}ria and
van den Heuvel, Henk and
Ide, Nancy and
Krek, Simon and
Toral, Antonio",
booktitle = "Proceedings of the Fifteenth Language Resources and Evaluation Conference",
month = may,
year = "2026",
address = "Palma de Mallorca, Spain",
publisher = "ELRA Language Resource Association",
url = "https://aclanthology.org/2026.lrec-1.791/",
doi = "10.63317/4rii4qtzbpew",
pages = "10082--10092",
abstract = "Detecting harmful content, such as cyberbullying, self-harm, and grooming, in self-generated content or conversations is an emerging research area with significant potential for positive social impact. However, challenges such as the scarcity of real-world conversational data, labor-intensive annotation processes, and inconsistent content policies hinder understanding and evaluating the performance of harmful content detection systems. In this study, we utilize openly available forum data to construct conversation proxies, facilitating the analysis and detection of harmful content. We undertook extensive efforts to label the conversational data using a consistent content policy developed by experts, with ten annotators contributing to the labeling process. Our experiments investigated the impact of context window size and found that performance in joint detection improved gradually up to a context window of 16 sentences, after which performance plateaued. Additionally, experiments with emojis demonstrated that using a tokenizer capable of decoding emojis yielded the best performance, while either removing emojis or converting them to text resulted in inferior outcomes."
}<?xml version="1.0" encoding="UTF-8"?>
<modsCollection xmlns="http://www.loc.gov/mods/v3">
<mods ID="hu-etal-2026-harnessing">
<titleInfo>
<title>Harnessing Synergy in Context and Emoji for Joint Detection of Harmful Online Content in Multi-turn Conversations</title>
</titleInfo>
<name type="personal">
<namePart type="given">Feiyan</namePart>
<namePart type="family">Hu</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Ciara</namePart>
<namePart type="given">Anne</namePart>
<namePart type="family">Byrne</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Jiang</namePart>
<namePart type="family">Zhou</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Rena</namePart>
<namePart type="family">Maycock</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Mark</namePart>
<namePart type="family">Langan</namePart>
<role>
<roleTerm authority="marcrelator" type="text">author</roleTerm>
</role>
</name>
<originInfo>
<dateIssued>2026-05</dateIssued>
</originInfo>
<typeOfResource>text</typeOfResource>
<relatedItem type="host">
<titleInfo>
<title>Proceedings of the Fifteenth Language Resources and Evaluation Conference</title>
</titleInfo>
<name type="personal">
<namePart type="given">Stelios</namePart>
<namePart type="family">Piperidis</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Núria</namePart>
<namePart type="family">Bel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Henk</namePart>
<namePart type="family">van den Heuvel</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Nancy</namePart>
<namePart type="family">Ide</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Simon</namePart>
<namePart type="family">Krek</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<name type="personal">
<namePart type="given">Antonio</namePart>
<namePart type="family">Toral</namePart>
<role>
<roleTerm authority="marcrelator" type="text">editor</roleTerm>
</role>
</name>
<originInfo>
<publisher>ELRA Language Resource Association</publisher>
<place>
<placeTerm type="text">Palma de Mallorca, Spain</placeTerm>
</place>
</originInfo>
<genre authority="marcgt">conference publication</genre>
</relatedItem>
<abstract>Detecting harmful content, such as cyberbullying, self-harm, and grooming, in self-generated content or conversations is an emerging research area with significant potential for positive social impact. However, challenges such as the scarcity of real-world conversational data, labor-intensive annotation processes, and inconsistent content policies hinder understanding and evaluating the performance of harmful content detection systems. In this study, we utilize openly available forum data to construct conversation proxies, facilitating the analysis and detection of harmful content. We undertook extensive efforts to label the conversational data using a consistent content policy developed by experts, with ten annotators contributing to the labeling process. Our experiments investigated the impact of context window size and found that performance in joint detection improved gradually up to a context window of 16 sentences, after which performance plateaued. Additionally, experiments with emojis demonstrated that using a tokenizer capable of decoding emojis yielded the best performance, while either removing emojis or converting them to text resulted in inferior outcomes.</abstract>
<identifier type="citekey">hu-etal-2026-harnessing</identifier>
<identifier type="doi">10.63317/4rii4qtzbpew</identifier>
<location>
<url>https://aclanthology.org/2026.lrec-1.791/</url>
</location>
<part>
<date>2026-05</date>
<extent unit="page">
<start>10082</start>
<end>10092</end>
</extent>
</part>
</mods>
</modsCollection>
%0 Conference Proceedings
%T Harnessing Synergy in Context and Emoji for Joint Detection of Harmful Online Content in Multi-turn Conversations
%A Hu, Feiyan
%A Byrne, Ciara Anne
%A Zhou, Jiang
%A Maycock, Rena
%A Langan, Mark
%Y Piperidis, Stelios
%Y Bel, Núria
%Y van den Heuvel, Henk
%Y Ide, Nancy
%Y Krek, Simon
%Y Toral, Antonio
%S Proceedings of the Fifteenth Language Resources and Evaluation Conference
%D 2026
%8 May
%I ELRA Language Resource Association
%C Palma de Mallorca, Spain
%F hu-etal-2026-harnessing
%X Detecting harmful content, such as cyberbullying, self-harm, and grooming, in self-generated content or conversations is an emerging research area with significant potential for positive social impact. However, challenges such as the scarcity of real-world conversational data, labor-intensive annotation processes, and inconsistent content policies hinder understanding and evaluating the performance of harmful content detection systems. In this study, we utilize openly available forum data to construct conversation proxies, facilitating the analysis and detection of harmful content. We undertook extensive efforts to label the conversational data using a consistent content policy developed by experts, with ten annotators contributing to the labeling process. Our experiments investigated the impact of context window size and found that performance in joint detection improved gradually up to a context window of 16 sentences, after which performance plateaued. Additionally, experiments with emojis demonstrated that using a tokenizer capable of decoding emojis yielded the best performance, while either removing emojis or converting them to text resulted in inferior outcomes.
%R 10.63317/4rii4qtzbpew
%U https://aclanthology.org/2026.lrec-1.791/
%U https://doi.org/10.63317/4rii4qtzbpew
%P 10082-10092
Markdown (Informal)
[Harnessing Synergy in Context and Emoji for Joint Detection of Harmful Online Content in Multi-turn Conversations](https://aclanthology.org/2026.lrec-1.791/) (Hu et al., LREC 2026)
ACL