<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="systematic-review" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1666349</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Systematic Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Cyberbullying detection approaches for Arabic texts: a systematic literature review</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Allwaibed</surname>
<given-names>Hooayda</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3135176/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Anbar</surname>
<given-names>Mohammed</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Manickam</surname>
<given-names>Selvakumar</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bintang</surname>
<given-names>Annisa</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Cybersecurity Research Centre (CYRES), Universiti Sains Malaysia (USM)</institution>, <addr-line>Penang</addr-line>, <country>Malaysia</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Computer Science, Applied College, Northern Border University</institution>, <addr-line>Arar</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff3"><sup>3</sup><institution>Universitas Indonesia Fakultas Ilmu Komputer</institution>, <addr-line>Depok</addr-line>, <country>Indonesia</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1434814/overview">Shadi Abudalfa</ext-link>, King Fahd University of Petroleum and Minerals, Saudi Arabia</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3076129/overview">Baligh Babaali</ext-link>, University of Medea, Algeria</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3105445/overview">Aadil Ganie</ext-link>, Universitat Polit&#x00E8;cnica de Val&#x00E8;ncia, Spain</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Hooayda Allwaibed, <email>Hooayda@student.usm.my</email>; Selvakumar Manickam, <email>selva@usm.my</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1666349</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>29</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Allwaibed, Anbar, Manickam and Bintang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Allwaibed, Anbar, Manickam and Bintang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>This study presents a comprehensive review of current methodologies, trends, and challenges in cyberbullying detection within Arabic-language contexts, with a focus on the unique linguistic and cultural factors associated with Arabic. This study reviews 35 peer-reviewed articles about the identification of cyberbullying in Arabic text. Reported accuracies across datasets and platforms range from approximately 73 to 96%, with precision frequently surpassing recall, suggesting that systems are more adept at identifying blatant bullying than at encompassing all pertinent instances. Methodologically, conventional machine learning utilizing Arabic-specific characteristics remains effective on smaller datasets, however deep neural architectures&#x2014;especially CNN/BiLSTM&#x2014;and transformer models like AraBERT yield superior outcomes when dialectal heterogeneity and orthographic noise are mitigated. Evaluation methodologies differ; research using a neutral class frequently indicates exaggerated accuracy, underscoring the necessity to emphasize macro-averaged F1 and per-class metrics. The evidence underscores deficiencies in dialectal representativeness, the uniformity of bullying notions compared to general abuse, and the transparency of annotation processes. Ethical and deployment considerations&#x2014;privacy preservation, dialectal bias, and real-time robustness&#x2014;are becoming increasingly significant. We integrate trends (models and features), standards (labeling and metrics), and future work directions, encompassing dialect-robust pretraining, cross-dataset evaluation, context-aware modeling, and human-in-the-loop frameworks. The review offers a comprehensive basis for researchers and practitioners pursuing culturally and linguistically tailored approaches to Arabic cyberbullying detection.</p>
</abstract>
<kwd-group>
<kwd>cyberbullying detection</kwd>
<kwd>Arabic language</kwd>
<kwd>systematic literature review</kwd>
<kwd>machine learning</kwd>
<kwd>deep learning</kwd>
<kwd>support vector machines</kwd>
<kwd>convolutional neural networks</kwd>
<kwd>natural language processing</kwd>
</kwd-group>
<counts>
<fig-count count="1"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="64"/>
<page-count count="13"/>
<word-count count="9029"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Natural Language Processing</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>The extensive utilization of digital communication channels has resulted in a concerning rise in cyberbullying, a type of online harassment impacting persons of many age groups and demographics. This study evaluated the relevant research published from 2014 to 2024, to assess and contrast the efficacy of conventional machine learning methods, deep learning frameworks, and sentiment-oriented strategies in the classification of cyberbullying, highlighting the significance of linguistic and dialectal intricacies in detection precision.</p>
<p>IT communication platforms such as WhatsApp, Facebook Messenger, Viber, WeChat, Line, Telegram, Imo, and Kakao Talk have increased in use throughout the last years, with some having over 1.5 billion users (<xref ref-type="bibr" rid="ref55">Urrutia Zubikarai, 2020</xref>). Several sources contended that offensive content in social media and communication platforms has become extremely dangerous; for instance, issues relating to social media in public institutions, particularly during the election period, are related to offensive content and have become challenging for public institutions in light of how information should be controlled (<xref ref-type="bibr" rid="ref34">Gr&#x00E9;goire et al., 2015</xref>). Offensive content, generally in the form of foul language spouting racial hate, personal attacks, and sexual harassment, is prevalent. Hence, it is important to detect offensive use of language to maintain a healthy discussion and enhance the security of users through the suppression of such hateful acts and offences (<xref ref-type="bibr" rid="ref24">Bertini et al., 2021</xref>; <xref ref-type="bibr" rid="ref46">Niraula et al., 2021</xref>). Online content-generators have increased, allowing more users to experience the freedom to express themselves, covered with anonymity if they choose, which maximizes the chance for platform misuse and leads to an environment that promotes offensive language and even eventually violence (<xref ref-type="bibr" rid="ref50">Sap et al., 2019</xref>). Also, social networking platforms display several types of offensive language like hate speech, aggressive content, cyberbullying, and toxic statements (<xref ref-type="bibr" rid="ref42">Miro&#x0144;czuk and Protasiewicz, 2018</xref>). A possible way to curtail and control such a phenomenon is through the use of NLP techniques like text classification for the automatic detection of offensive language. More specifically, text classification is the process of labelling new text with pre-defined labels (<xref ref-type="bibr" rid="ref42">Miro&#x0144;czuk and Protasiewicz, 2018</xref>).</p>
</sec>
<sec id="sec2">
<label>2</label>
<title>Background of study</title>
<sec id="sec3">
<label>2.1</label>
<title>Cyberbullying</title>
<p>Cyberbullying has become a global concern with the rise of social media and online platforms, and research efforts are increasingly being devoted to detecting and mitigating it using Machine Learning (ML), Deep Learning (DL), and Natural Language Processing (NLP) approaches. While a significant amount of research has been conducted in languages like English, studies targeting cyberbullying in Arabic remain limited. This systematic literature review aims to explore existing research on cyberbullying detection in the Arabic language, with a focus on ML and DL techniques, and to identify future research directions based on the analysis of the reviewed studies.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Challenges in detecting in Arabic language</title>
<p>Identifying cyberbullying in the Arabic language poses difficulties, mostly due to the linguistic, cultural, and computational intricacies involved in processing Arabic content. A principal challenge is the significant range of Arabic dialects, which differ not only by area but also by socio-economic and cultural factors. Although Modern Standard Arabic (MSA) is extensively employed in formal discourse, social media exchanges primarily transpire in dialectal Arabic, which is characterized by the absence of standardized spelling, syntax, and vocabulary (<xref ref-type="bibr" rid="ref45">Mubarak and Darwish, 2019</xref>; <xref ref-type="bibr" rid="ref1">AbdelHamid et al., 2022</xref>). The lack of high-quality, labeled datasets that consider these changes intensifies the issue, resulting in diminished model performance in real-world Arabic cyberbullying detection tasks (<xref ref-type="bibr" rid="ref23">Bashir and Bouguessa, 2021</xref>; <xref ref-type="bibr" rid="ref40">Khairy et al., 2023</xref>). A fundamental problem is the morphological complexity and intricate syntax of Arabic, which markedly contrasts with Indo-European languages like English. Arabic lexicon demonstrates significant inflexion through affixation, root-based derivations, and contextual variants, complicating tokenization, stemming, and lemmatization (<xref ref-type="bibr" rid="ref6">Alakrot et al., 2018</xref>; <xref ref-type="bibr" rid="ref9008">Haidar et al., 2019</xref>). The linguistic features create difficulty in text classification, as identical words may possess varying meanings based on diacritical marks, which are frequently absent in informal online communication. The scarcity of comprehensive pre-trained models tailored for Arabic dialects constrains the capacity of NLP algorithms to effectively identify harmful and abusive content (<xref ref-type="bibr" rid="ref14">Alrashidi et al., 2023</xref>; <xref ref-type="bibr" rid="ref41">Khezzar et al., 2023</xref>). Research indicates that sentiment analysis and lexicon-based methodologies can improve detection by identifying emotional indicators; however, their efficacy is limited by the necessity for manually curated lexicons specific to Arabic dialects (<xref ref-type="bibr" rid="ref33">Farid and El-Tazi, 2020</xref>). An application of NLP that extracts structured information in the form of entities, entities&#x2019; relationship and attributes describing them from unstructured documents in an automatic method is Information Extraction (IE) (<xref ref-type="bibr" rid="ref30">Cowie and Lehnert, 1996</xref>). Besides, IE systems have been found effective in handling information overload issues, enabling the discernment of the most significant information portion from a huge portion of information in a timely and easy manner. On the whole, detection of offensive language online is possible through the development of a model using ML, AI, DL and NLP methods. This paper investigates the following research questions:</p>
</sec>
</sec>
<sec id="sec5">
<label>3</label>
<title>Research questions</title>
<disp-quote>
<p><bold>Q1:</bold> What are the current trends in cyberbullying detection for the Arabic language and which dialects do they cover?</p>
</disp-quote>
<disp-quote>
<p><bold>Q2:</bold> How cyberbullying been detected in previous studies based on standards that represent its definition and characteristics?</p>
</disp-quote>
<disp-quote>
<p><bold>Q3:</bold> What directions for future research in cyberbullying detection may be established based on the findings of this review?</p>
</disp-quote>
</sec>
<sec sec-type="methods" id="sec6">
<label>4</label>
<title>Methodology</title>
<p>A systematic literature review was conducted to conduct a comprehensive analysis by focusing on existing studies from 2014 to 2024, evaluating trends and advancements in cyberbullying detection for Arabic texts. This methodology involves structured selection criteria to ensure that only relevant and high-quality sources are included. The Inclusion Criteria are as follows:</p>
<list list-type="order">
<list-item>
<p>Studies published from 2014 to 2024</p>
</list-item>
<list-item>
<p>Articles in English</p>
</list-item>
<list-item>
<p>Research specific to Arabic text-based cyberbullying detection</p>
</list-item>
</list>
<p>The exclusion criteria were:</p>
<list list-type="order">
<list-item>
<p>The research focused on social studies without technological elements</p>
</list-item>
<list-item>
<p>Studies in languages other than English and non-Arabic texts</p>
</list-item>
<list-item>
<p>Non-text-based detection methods (e.g., voice, image, video)</p>
</list-item>
<list-item>
<p>Conference papers and review articles</p>
</list-item>
</list>
<p>SLR protocol was applied to the study, the final selected studies were conducted, and theoretical and practical steps were taken while conducting the SLR.</p>
</sec>
<sec id="sec7">
<label>5</label>
<title>Data sources and keywords</title>
<p>In the first step, four major research databases, ScienceDirect, Scopus, Web of Science, and Springer, were searched through queries, and as many papers as possible were collected. The search query is &#x201C;detect&#x201D; AND (&#x201C;cyberbullying&#x201D; OR &#x201C;hate speech&#x201D; OR &#x201C;harassment&#x201D; OR &#x201C;offensive&#x201D;) AND (&#x201C;machine learning&#x201D; OR &#x201C;natural language processing&#x201D; OR &#x201C;deep learning&#x201D;) AND &#x201C;Arabic.&#x201D; Based on initial exclusion criteria, papers were selected after carefully reading the abstracts of the papers in the second step. A final list of papers is prepared after reading the full articles and applying further exclusion criteria (35 papers). <xref ref-type="fig" rid="fig1">Figure 1</xref> depicts the literature review process.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Systematic literature review process.</p>
</caption>
<graphic xlink:href="frai-08-1666349-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Flowchart illustrating study identification via databases and registers. initially, 837 records were screened, resulting in 516 being excluded. All 321 of the remaining reports were successfully retrieved for eligibility assessment. Of these, 186 reports were excluded, leaving 35 studies for the final review.</alt-text>
</graphic>
</fig>
</sec>
<sec sec-type="results" id="sec8">
<label>6</label>
<title>Results</title>
<p>This review synthesizes findings from numerous studies on cyberbullying detection within Arabic-language content, identifying the main trends, challenges, and methodologies, including ML, DL, and sentiment analysis. The majority of the studies concentrated on cyberbullying detection, offensive language detection, and hate speech identification. A significant portion of the research applied to social media platforms like Twitter and YouTube. The focus was largely on identifying cyberbullying in dialects such as Saudi Arabian Arabic, Egyptian Arabic, and the Levantine dialects. The most frequently used machine learning models included Na&#x00EF;ve Bayes (NB), Support Vector Machine (SVM), and Random Forest (RF). For deep learning models, LSTM, CNN, and GRU were prominent. Ensemble techniques like stacking and boosting showed better performance compared to individual ML models. The datasets used in the reviewed studies varied widely in size, ranging from small manually annotated datasets to large datasets collected from social media. Many studies employed preprocessing techniques such as tokenization, stemming, lemmatization, and removal of hyperlinks or non-Arabic characters to clean the data before analysis. Preprocessing was critical in ensuring the effectiveness of the ML and DL models. Across the reviewed studies, model performance is generally strong, with traditional machine learning and deep learning approaches demonstrating reliable detection capabilities in Arabic cyberbullying contexts. Reported results indicate that precision commonly exceeds recall, suggesting that systems are better at correctly identifying bullying instances than capturing all relevant cases. This pattern appears in works employing classical classifiers as well as ensemble strategies, with examples including Egyptian-dialect tweet classification (<xref ref-type="bibr" rid="ref33">Farid and El-Tazi, 2020</xref>), Na&#x00EF;ve Bayes&#x2013;based detection pipelines (<xref ref-type="bibr" rid="ref44">Mouheb et al., 2019</xref>), offensive language identification in user-generated video comments (<xref ref-type="bibr" rid="ref6">Alakrot et al., 2018</xref>), and ensemble machine learning frameworks that optimize the balance of precision and recall (<xref ref-type="bibr" rid="ref9008">Haidar et al., 2019</xref>). In terms of offensive language and cyberbullying detection, researchers identify various types of offensive language, each reflecting specific social, cultural, and regional sensitivities. <xref ref-type="table" rid="tab1">Table 1</xref> illustrates the types of offensive language used in Arabic studies on cyberbullying and offensive content</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Types of offensive language used in Arabic studies on cyberbullying and offensive content.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Type of Offensive Language</th>
<th align="left" valign="top">Description</th>
<th align="left" valign="top">Sources</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Hate Speech</td>
<td align="left" valign="top">Language targeting specific groups based on religion, race, gender, or nationality. Includes:</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref29">Casta&#x00F1;o-Pulgar&#x00ED;n et al. (2021)</xref>, <xref ref-type="bibr" rid="ref15">Alsafari et al. (2020a</xref>, <xref ref-type="bibr" rid="ref16">2020b)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Insults and Personal Attacks</td>
<td align="left" valign="top">Abusive language aimed at degrading individuals, including name-calling, derogatory remarks, and personal insults about appearance, intelligence, or social status.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref17">Alshalabi et al. (2024)</xref>,</td>
</tr>
<tr>
<td align="left" valign="top">Profanity and Vulgar Language</td>
<td align="left" valign="top">Taboo words or phrases generally considered offensive, including swear words and obscenities that are often censored on public platforms.</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref49">Rosenbaum (2019)</xref>
</td>
</tr>
<tr>
<td align="left" valign="top">Sexual Harassment</td>
<td align="left" valign="top">Inappropriate comments or sexually explicit content targeting individuals, often related to gender-based discrimination.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref2">Abdelmonem (2015)</xref>, <xref ref-type="bibr" rid="ref25">Bouhlila (2019)</xref>, <xref ref-type="bibr" rid="ref24">Bertini et al. (2021)</xref>, <xref ref-type="bibr" rid="ref46">Niraula et al. (2021)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Bullying and Harassment</td>
<td align="left" valign="top">Repeated or persistent offensive behavior aimed at intimidating or humiliating someone, often through derogatory remarks about personal life or achievements.</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref38">Kanan et al. (2020)</xref>
</td>
</tr>
<tr>
<td align="left" valign="top">Stereotyping and Discrimination</td>
<td align="left" valign="top">Generalizations that promote negative stereotypes about specific groups (e.g., based on age, nationality, profession). Includes implicit bias and discriminatory remarks.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref15">Alsafari et al. (2020a</xref>, <xref ref-type="bibr" rid="ref16">2020b)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Mockery and Sarcasm</td>
<td align="left" valign="top">Humorous or sarcastic language used to belittle or degrade individuals or groups, often through irony or exaggeration, which can vary in offensiveness depending on context.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref3">Abu Farha (2023)</xref>.</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="sec9">
<label>6.1</label>
<title>Research question 1</title>
<p>The first research question was:</p>
<p>What are the current trends in cyberbullying detection for the Arabic language, and how do these trends account for various dialects?</p>
<p>The following themes were developed to answer the first research question 1:</p>
<sec id="sec10">
<label>6.1.1</label>
<title>Machine learning (ML) and deep learning (DL) approaches</title>
<p>Several studies have utilized ML and DL algorithms to detect cyberbullying, with Support Vector Machine (SVM) and Na&#x00EF;ve Bayes (NB) being frequently applied (e.g., <xref ref-type="bibr" rid="ref35">Haidar et al., 2017</xref>; <xref ref-type="bibr" rid="ref6">Alakrot et al., 2018</xref>). More recently, DL methods such as Convolutional Neural Networks (CNNs) and Recurrent Neural Networks (RNNs) have demonstrated improved performance due to their ability to capture context and semantic meanings in text (e.g., <xref ref-type="bibr" rid="ref36">Haidar et al., 2018</xref>; <xref ref-type="bibr" rid="ref44">Mouheb et al., 2019</xref>; <xref ref-type="bibr" rid="ref43">Mohaouchane et al., 2019</xref>). Ensemble learning, where multiple models are combined to improve prediction accuracy, has shown promise in boosting performance. For instance, stacking, boosting, and bagging techniques have demonstrated better performance in detecting Arabic cyberbullying content (e.g., <xref ref-type="bibr" rid="ref36">Haidar et al., 2018</xref>; <xref ref-type="bibr" rid="ref40">Khairy et al., 2023</xref>; <xref ref-type="table" rid="tab2">Table 2</xref>).</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Summary of reviewed studies on Arabic hate/offensive/cyberbullying detection.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">No</th>
<th align="left" valign="top">Study</th>
<th align="left" valign="top">Model(s)</th>
<th align="left" valign="top">Dataset and Platform</th>
<th align="left" valign="top">Dialect/Domain</th>
<th align="left" valign="top">Performance Metrics</th>
<th align="left" valign="top">Limitations</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">1</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref35">Haidar et al. (2017)</xref>
</td>
<td align="left" valign="top">Na&#x00EF;ve Bayes, SVM</td>
<td align="left" valign="top">Posts (Twitter, Facebook, Formspring)</td>
<td align="left" valign="top">Saudi Arabic</td>
<td align="left" valign="top">NB: Precision 90.85%; SVM: Precision 0.815 (yes class)</td>
<td align="left" valign="top">Imbalanced dataset; few bullying instances; precision misleading</td>
</tr>
<tr>
<td align="left" valign="top">2</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref36">Haidar et al. (2018)</xref>
</td>
<td align="left" valign="top">Feed-forward Neural Network (DL)</td>
<td align="left" valign="top">Twitter dataset (binary labels)</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">Validation accuracy 91.17% (7 hidden layers)</td>
<td align="left" valign="top">Limited to binary labels; dataset size not large</td>
</tr>
<tr>
<td align="left" valign="top">3</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref6">Alakrot et al. (2018)</xref>
</td>
<td align="left" valign="top">SVM</td>
<td align="left" valign="top">YouTube comments</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">Precision 90.05%</td>
<td align="left" valign="top">Small dataset; not specific to cyberbullying</td>
</tr>
<tr>
<td align="left" valign="top">4</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref9">AlHarbi et al. (2019)</xref>
</td>
<td align="left" valign="top">Lexicon + Sentiment Analysis (PMI, Chi-square, Entropy)</td>
<td align="left" valign="top">Tweets</td>
<td align="left" valign="top">Twitter (Saudi Arabic)</td>
<td align="left" valign="top">PMI accuracy 81% vs. Chi-square 62.11%</td>
<td align="left" valign="top">Lexicon-based; potential bias; dataset context-limited</td>
</tr>
<tr>
<td align="left" valign="top">5</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref45">Mubarak and Darwish (2019)</xref>
</td>
<td align="left" valign="top">ML classifiers</td>
<td align="left" valign="top">Arabic tweets</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">High classification accuracy</td>
<td align="left" valign="top">Focused only on offensive, not cyberbullying</td>
</tr>
<tr>
<td align="left" valign="top">6</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref33">Farid and El-Tazi (2020)</xref>
</td>
<td align="left" valign="top">Lexicon-based Sentiment Analysis + Emojis</td>
<td align="left" valign="top">Tweets in Modern Standard + Egyptian Dialect</td>
<td align="left" valign="top">Egyptian Arabic</td>
<td align="left" valign="top">Accuracy &#x003E;73% for bullying hashtags</td>
<td align="left" valign="top">Lexicon limited; reliance on emojis and history</td>
</tr>
<tr>
<td align="left" valign="top">7</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref16">Alsafari et al. (2020b)</xref>
</td>
<td align="left" valign="top">LR, LSTM, Sluice, BERT, ELMo, SVM</td>
<td align="left" valign="top">Labeled tweets (Twitter)</td>
<td align="left" valign="top">Mixed Arabic dialects</td>
<td align="left" valign="top">SVM&#x202F;+&#x202F;ngrams: Acc. 85.16%; CNN&#x202F;+&#x202F;mBERT F1-macro 66.86%</td>
<td align="left" valign="top">Limited samples per class; subjectivity in annotation</td>
</tr>
<tr>
<td align="left" valign="top">8</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref23">Bashir and Bouguessa (2021)</xref>
</td>
<td align="left" valign="top">LSTM, SVM, Na&#x00EF;ve Bayes</td>
<td align="left" valign="top">Twitter dataset (cyberbullying keywords)</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">LSTM accuracy 72%</td>
<td align="left" valign="top">Keyword-based data collection; lower accuracy</td>
</tr>
<tr>
<td align="left" valign="top">9</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref9007">Fati (2022)</xref>
</td>
<td align="left" valign="top">Sentiment Analysis Framework</td>
<td align="left" valign="top">Twitter</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">Accuracy 81% (10-fold CV)</td>
<td align="left" valign="top">Limited validation; binary annotation</td>
</tr>
<tr>
<td align="left" valign="top">10</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref10">Al-Hassan and Al-Dossari (2022)</xref>
</td>
<td align="left" valign="top">LSTM, CNN&#x202F;+&#x202F;LSTM, GRU, CNN&#x202F;+&#x202F;GRU</td>
<td align="left" valign="top">Labeled tweets</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">CNN&#x202F;+&#x202F;LSTM: Precision 72%, Recall 75%, F1 73%</td>
<td align="left" valign="top">Moderate dataset size; limited categories</td>
</tr>
<tr>
<td align="left" valign="top">11</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref18">Alsubait and Alfageh (2021)</xref>
</td>
<td align="left" valign="top">Multinomial NB, Complement NB, Logistic Regression</td>
<td align="left" valign="top">YouTube comments</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">Avg. F1: TF-IDF 77.9% vs. CountVec 77.5%</td>
<td align="left" valign="top">Dataset modest; no deep learning comparison</td>
</tr>
<tr>
<td align="left" valign="top">12</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref9003">Alhashmi and Darem (2022)</xref>
</td>
<td align="left" valign="top">RF, NB, SVM, XGB, ANN, Stacked DL; Consensus-Based Ensemble</td>
<td align="left" valign="top">(Twitter, WhatsApp, Vine, Instagram, Packet; incl. Translated data)</td>
<td align="left" valign="top">Mixed Arabic + translated</td>
<td align="left" valign="top">Consensus ensemble improved accuracy by 1.3% over best classifier; RF strongest</td>
<td align="left" valign="top">Dataset partly translated; mixed domains; modest gain over baselines</td>
</tr>
<tr>
<td align="left" valign="top">13</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref9006">Bouliche and Rezoug (2022)</xref>
</td>
<td align="left" valign="top">Dynamic Graph Neural Network (DGNN)</td>
<td align="left" valign="top">Arabic comments (tweets)</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">Accuracy 74%</td>
<td align="left" valign="top">Model performance modest; needs refinement; small dataset</td>
</tr>
<tr>
<td align="left" valign="top">14</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref31">El-Alami et al. (2022)</xref>
</td>
<td align="left" valign="top">BERT (multilingual, transfer learning)</td>
<td align="left" valign="top">Bilingual dataset (English + Arabic tweets)</td>
<td align="left" valign="top">General Arabic + English</td>
<td align="left" valign="top">High accuracy and F1; BERT outperformed other models</td>
<td align="left" valign="top">Ambiguous language still difficult; early-stage</td>
</tr>
<tr>
<td align="left" valign="top">15</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref1">AbdelHamid et al. (2022)</xref>
</td>
<td align="left" valign="top">AraBERT, ArabicBERT, GigaBERT vs. RF, SVM</td>
<td align="left" valign="top">Syrian/Levantine tweets</td>
<td align="left" valign="top">Levantine dialect</td>
<td align="left" valign="top">GigaBERT: AUC 94.6%, Macro F1 0.81</td>
<td align="left" valign="top">Focused on Levantine; dataset scope limited</td>
</tr>
<tr>
<td align="left" valign="top">16</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref8">AlFarah et al. (2022)</xref>
</td>
<td align="left" valign="top">SVM, RF, NB, LR, KNN</td>
<td align="left" valign="top">Twitter + YouTube, oversampled</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">NB highest AUC 89%; SVM and LR also strong</td>
<td align="left" valign="top">Class imbalance; dataset moderate in size</td>
</tr>
<tr>
<td align="left" valign="top">17</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref21">Anezi (2022)</xref>
</td>
<td align="left" valign="top">Deep Recurrent Neural Network (DRNN)</td>
<td align="left" valign="top">Custom Arabic comments dataset</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">Binary Acc 99.73%; 3-class Acc 95.38%; 7-class Acc 84.14%</td>
<td align="left" valign="top">Dataset unique but limited disclosure; overfitting risk</td>
</tr>
<tr>
<td align="left" valign="top">18</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref19">Althobaiti (2022)</xref>
</td>
<td align="left" valign="top">BERT + Sentiment + Emoji features vs. SVM, LR</td>
<td align="left" valign="top">Arabic tweets</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">BERT model highest F1 across all tasks</td>
<td align="left" valign="top">Single dataset; limited external validation</td>
</tr>
<tr>
<td align="left" valign="top">19</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref11">Ali and Kurdy (2022)</xref>
</td>
<td align="left" valign="top">SVM, SGD, KNN, LR, AdaBoost, Bagging</td>
<td align="left" valign="top">Syrian Facebook comments + questionnaire</td>
<td align="left" valign="top">Syrian slang</td>
<td align="left" valign="top">SVM and SGD accuracy 77%; AdaBoost precision 94%</td>
<td align="left" valign="top">Imbalanced recall (47%); small dataset</td>
</tr>
<tr>
<td align="left" valign="top">20</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref9002">Alduailaj and Belghith (2023)</xref>
</td>
<td align="left" valign="top">SVM&#x202F;+&#x202F;FarasaNLTK vs. NB</td>
<td align="left" valign="top">Twitter + YouTube comments</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">SVM best accuracy 95.74% (TF-IDF n-gram)</td>
<td align="left" valign="top">Keyword-based collection; possible bias</td>
</tr>
<tr>
<td align="left" valign="top">21</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref40">Khairy et al. (2023)</xref>
</td>
<td align="left" valign="top">Ensemble (Voting) vs. LR, SVC, KNN</td>
<td align="left" valign="top">New balanced dataset</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">Voting model highest Acc, F1, Recall, Precision; LR best single Acc 65.1%</td>
<td align="left" valign="top">Small dataset; limited to ML</td>
</tr>
<tr>
<td align="left" valign="top">22</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref48">Rachidi et al. (2023)</xref>
</td>
<td align="left" valign="top">ML (SVM, NB, RF, LR) and DL (LSTM)</td>
<td align="left" valign="top">Instagram Moroccan dialect</td>
<td align="left" valign="top">Moroccan Arabic</td>
<td align="left" valign="top">LSTM Acc 83.64%; SVM Acc 75.04%</td>
<td align="left" valign="top">Scarcity of tools/datasets for dialect; modest results</td>
</tr>
<tr>
<td align="left" valign="top">23</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref14">Alrashidi et al. (2023)</xref>
</td>
<td align="left" valign="top">Fine-tuned Arabic BERT, Multi-task Learning</td>
<td align="left" valign="top">Multi-aspect abusive tweets dataset</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">MTL&#x202F;+&#x202F;BERT &#x003E; DL baselines; GitHub data shared</td>
<td align="left" valign="top">Imbalanced datasets; Arabic only</td>
</tr>
<tr>
<td align="left" valign="top">24</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref32">Elzayady et al. (2023)</xref>
</td>
<td align="left" valign="top">CNN-LSTM, CNN-BiLSTM, CNN-GRU, AraBERT +Personality Features</td>
<td align="left" valign="top">Twitter hate speech dataset</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">AraBERT + personality features Acc 82.3%; CNN-LSTM 77%</td>
<td align="left" valign="top">Personality inference adds complexity; dataset size moderate</td>
</tr>
<tr>
<td align="left" valign="top">25</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref41">Khezzar et al. (2023)</xref>
</td>
<td align="left" valign="top">LR, SVC, DT, CNN, AraBERT; web app (arHateDetector)</td>
<td align="left" valign="top">arHateDataset (merged public sets), Twitter</td>
<td align="left" valign="top">Standard + dialectal Arabic</td>
<td align="left" valign="top">AraBERT accuracy 93%; precision/recall/F1 reported</td>
<td align="left" valign="top">Aggregated datasets may introduce label/definition drift; external validation not detailed</td>
</tr>
<tr>
<td align="left" valign="top">26</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref15">Alsafari et al. (2020a)</xref>
</td>
<td align="left" valign="top">Single and ensemble CNN/BiLSTM; AraBERT vs. non-contextual embeddings</td>
<td align="left" valign="top">Twitter; fine-grained two/three/six-class corpora</td>
<td align="left" valign="top">Mixed Arabic dialects</td>
<td align="left" valign="top">Ensemble F1: 91% (2-class), 84% (3-class), 80% (6-class); AraBERT &#x003E; non-contextual; CNN&#x202F;&#x003E;&#x202F;BiLSTM</td>
<td align="left" valign="top">Class granularity increases difficulty; error analysis shows issues with implicit/defensive language</td>
</tr>
<tr>
<td align="left" valign="top">27</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref13">Aljuhani et al. (2022)</xref>
</td>
<td align="left" valign="top">BiLSTM with domain-specific embeddings; LR, SVM baselines</td>
<td align="left" valign="top">Tweets (seeded crawl, cleaned, labeled)</td>
<td align="left" valign="top">General Arabic (Twitter)</td>
<td align="left" valign="top">LR on char n-grams P/R/F1&#x202F;=&#x202F;92%; SVM&#x202F;&#x2248;&#x202F;90%; BiLSTM competitive with domain embeddings</td>
<td align="left" valign="top">Seed-term collection bias; translation/generalization across topics not assessed</td>
</tr>
<tr>
<td align="left" valign="top">28</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref20">Amer Hamzah and Dhannoon (2023)</xref>
</td>
<td align="left" valign="top">BiLSTM + Temporal Convolutional Network (TCN)</td>
<td align="left" valign="top">CASH: tweets on sexual harassment</td>
<td align="left" valign="top">Sexual-harassment domain</td>
<td align="left" valign="top">Accuracy 96.65%; F0.5&#x202F;=&#x202F;0.969; &#x003E; XGBoost baseline</td>
<td align="left" valign="top">Task/domain specific; dialectal robustness not analyzed</td>
</tr>
<tr>
<td align="left" valign="top">29</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref26">Boulouard et al. (2022)</xref>
</td>
<td align="left" valign="top">BERT EN, AraBERT, mBERT (AR/EN), LinearSVC, LSTM</td>
<td align="left" valign="top">YouTube comments (Gulf, Egyptian, Iraqi); Tweets</td>
<td align="left" valign="top">Mixed Arabic dialects; EN translations</td>
<td align="left" valign="top">BERT EN Acc 98%; AraBERT Acc 96%; mBERT-AR Acc 83%; LSTM Acc 82%</td>
<td align="left" valign="top">Translation pipeline may inflate EN results; sarcasm remains challenging</td>
</tr>
<tr>
<td align="left" valign="top">30</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref9004">Aljarah et al. (2021)</xref>
</td>
<td align="left" valign="top">SVM, NB, DT, RF; feature sets (TF-IDF, profile, emotion)</td>
<td align="left" valign="top">Twitter</td>
<td align="left" valign="top">General Arabic (varied topics)</td>
<td align="left" valign="top">RF best: Acc/G-mean 0.910; Recall 0.923; Precision 0.902 with all features</td>
<td align="left" valign="top">Small corpus after filtering; two-annotator protocol; neutrals excluded from training</td>
</tr>
<tr>
<td align="left" valign="top">31</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref44">Mouheb et al. (2019)</xref>
</td>
<td align="left" valign="top">Na&#x00EF;ve Bayes</td>
<td align="left" valign="top">Twitter + YouTube</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">Accuracy 0.959</td>
<td align="left" valign="top">Small dataset; limited feature diversity</td>
</tr>
<tr>
<td align="left" valign="top">32</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref5">Alakrot et al. (2021)</xref>
</td>
<td align="left" valign="top">LR, SVM/LinearSVC, NB, DT, RF; POS&#x202F;+&#x202F;n-grams; feature selection</td>
<td align="left" valign="top">YouTube comments</td>
<td align="left" valign="top">Mixed dialects (YouTube)</td>
<td align="left" valign="top">LinearSVC highest accuracy (reasonable overall); gains from feature selection</td>
<td align="left" valign="top">Focus on offensive, not CB; dependence on preprocessing choices</td>
</tr>
<tr>
<td align="left" valign="top">33</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref47">Omar et al. (2021)</xref>
</td>
<td align="left" valign="top">LinearSVC, NB variants, SVM, LR, DT, SGD, RF; multilabel pipeline</td>
<td align="left" valign="top">OSN posts across 11 classes; vulgar-speech set</td>
<td align="left" valign="top">General Arabic (Facebook/Twitter)</td>
<td align="left" valign="top">With Chi-square FS: Acc 97.92%; F1 97.92%; Precision 97.92%; Recall 97.93%; multilabel LinearSVC + TF-IDF Acc 82.29%, F1 92.48%</td>
<td align="left" valign="top">High feature counts; results sensitive to FS; generalizability outside OSN mix not shown</td>
</tr>
<tr>
<td align="left" valign="top">34</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref53">Shannaq et al. (2022)</xref>
</td>
<td align="left" valign="top">Word-embedding fine-tuning + GA-optimized SVM/XGBoost</td>
<td align="left" valign="top">ArCybC (CB/Non-CB/Off/Non-Off)</td>
<td align="left" valign="top">Twitter; cyberbullying + offensive</td>
<td align="left" valign="top">SVM Acc 86.5%&#x202F;&#x2192;&#x202F;87.5%; XGB Acc 84.9%&#x202F;&#x2192;&#x202F;85.2% after optimization</td>
<td align="left" valign="top">Incremental gains; relies on a single public corpus</td>
</tr>
<tr>
<td align="left" valign="top">35</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref39">Kanan et al. (2021)</xref>
</td>
<td align="left" valign="top">Unsupervised K-Means vs. EM (clustering)</td>
<td align="left" valign="top">(Facebook/Twitter)</td>
<td align="left" valign="top">General Arabic</td>
<td align="left" valign="top">Evaluated via training time, SSE (e.g., 7,796.363), and log-likelihood (e.g., 3,606.4669)</td>
<td align="left" valign="top">No precision/recall/F1; clustering quality hard to align with downstream moderation needs</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec11">
<label>6.1.2</label>
<title>Sentiment analysis and lexicon-based methods</title>
<p>Sentiment analysis, often coupled with lexicon-based approaches, is commonly used to detect harmful content. <xref ref-type="bibr" rid="ref9">AlHarbi et al. (2019)</xref> and <xref ref-type="bibr" rid="ref33">Farid and El-Tazi (2020)</xref> used sentiment-based lexicons for Arabic texts, finding that pointwise mutual information (PMI) and lexicon enhancement can improve detection accuracy. Sentiment-based approaches are also utilized alongside NLP tools, such as tokenization and stemming, for feature extraction, enhancing the ability to detect cyberbullying based on emotional cues.</p>
</sec>
<sec id="sec12">
<label>6.1.3</label>
<title>Handling Arabic dialects and linguistic complexity</title>
<p>Dialectal Arabic presents a significant challenge, as standard ML models may not perform well on diverse dialects. Studies such as <xref ref-type="bibr" rid="ref18">Alsubait and Alfageh (2021)</xref> and <xref ref-type="bibr" rid="ref10">Al-Hassan and Al-Dossari (2022)</xref> indicate that datasets tailored to specific dialects (e.g., Egyptian, Levantine) enhance detection efficacy. Additionally, transformer-based models like AraBERT and multilingual BERT have emerged as effective tools for dealing with dialectal variations, as they can better capture semantic nuances across dialects (e.g., <xref ref-type="bibr" rid="ref41">Khezzar et al., 2023</xref>; <xref ref-type="bibr" rid="ref14">Alrashidi et al., 2023</xref>).</p>
</sec>
</sec>
<sec id="sec13">
<label>6.2</label>
<title>Research question 2</title>
<p>How has cyberbullying been detected in previous studies based on standards that represent its definition and characteristics?</p>
<p>The following themes were developed to answer the second research question.</p>
<sec id="sec14">
<label>6.2.1</label>
<title>Development and use of cyberbullying datasets</title>
<p>Arabic cyberbullying detection relies heavily on curated datasets. Studies often use platform-specific datasets from Twitter, YouTube, and Facebook, with datasets labeled for harmful or offensive language (e.g., <xref ref-type="bibr" rid="ref23">Bashir and Bouguessa, 2021</xref>; <xref ref-type="bibr" rid="ref40">Khairy et al., 2023</xref>). These datasets include common cyberbullying characteristics like threats, insults, and hate speech. However, the issue of dataset imbalance (more non-cyberbullying content than cyberbullying) persists, affecting model performance. Techniques like oversampling and downsampling have been employed to address this imbalance, as seen in <xref ref-type="bibr" rid="ref8">AlFarah et al. (2022)</xref>. <xref ref-type="table" rid="tab3">Table 3</xref>. Shows some examples of the existing datasets addressing cyberbullying in Arabic.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Examples of the datasets addressing cyberbullying in Arabic.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Dataset (year)</th>
<th align="left" valign="top">Platform</th>
<th align="left" valign="top">Labels</th>
<th align="left" valign="top">Study</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Instagram-Based Benchmark Dataset for Cyberbullying in Arabic (2022)</td>
<td align="left" valign="top">Instagram</td>
<td align="left" valign="top">Comments collected; multi-class sub-categories for bullying with sentiment variants used in evaluation (incl. Positive/negative/neutral)</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref7">Albayari and Abdallah (2022)</xref>
</td>
</tr>
<tr>
<td align="left" valign="top">ArCybC / ArCyC&#x2014;Arabic Cyberbullying Corpus (2022 article; 2023 data release)</td>
<td align="left" valign="top">Twitter (X)</td>
<td align="left" valign="top">Tweets; dual annotation tasks: CB vs. non-CB and Offensive vs. non-Offensive; 5 annotators</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref52">Shannag et al. (2022)</xref>
</td>
</tr>
<tr>
<td align="left" valign="top">ArbCyD&#x2014;Arabic Post Dataset for Cyberbullying Detection (2024)</td>
<td align="left" valign="top">Twitter (X)</td>
<td align="left" valign="top">Posts: bullying vs. non-bullying binary labels</td>
<td align="left" valign="top">
<xref ref-type="bibr" rid="ref12">Aljalaoud et al. (2025)</xref>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The ArCybC/ArCyC corpus represents one of the few openly accessible multi-dialect Twitter datasets that makes a clear distinction between cyberbullying and general offensive content. Its development is supported by detailed documentation of the annotation pipeline and guidelines, ensuring methodological transparency (<xref ref-type="bibr" rid="ref52">Shannag et al., 2022</xref>). The ArbCyD dataset significantly expands the available volume by including annotated Twitter posts (<xref ref-type="bibr" rid="ref12">Aljalaoud et al., 2025</xref>).</p>
</sec>
<sec id="sec15">
<label>6.2.2</label>
<title>Standards and evaluation metrics</title>
<p>Standards such as precision, recall, F1-score, and accuracy are commonly used to evaluate detection methods (e.g., <xref ref-type="bibr" rid="ref35">Haidar et al., 2017</xref>; <xref ref-type="bibr" rid="ref6">Alakrot et al., 2018</xref>). Although precision and recall are essential for accurate detection, the unique characteristics of the Arabic language and cyberbullying-specific terms often require additional metrics and customized standards. Studies such as <xref ref-type="bibr" rid="ref31">El-Alami et al. (2022)</xref> and <xref ref-type="bibr" rid="ref20">Amer Hamzah and Dhannoon (2023)</xref> advocate for using contextual features like sentiment polarity, emojis, and user history in cyberbullying detection. These standards help capture the nuanced characteristics of online abuse, especially within specific platforms or dialects.</p>
<p>Some evaluations adopt three-way labeling schemes that distinguish bullying/abusive content, non-bullying content, and neutral content. When overall accuracy is computed across all classes, the typically high prevalence of neutral instances can inflate the metric and obscure a system&#x2019;s effectiveness on the bullying class, which is the primary target in safety-critical applications. For example, the Instagram-based Arabic cyberbullying benchmark provides a multi-class design with positive (bullying), negative (non-bullying), and neutral categories, together with inter-annotator agreement reporting and baseline models (<xref ref-type="bibr" rid="ref7">Albayari and Abdallah, 2022</xref>). In such settings, macro-F1 and per-class F1 are preferable for comparing systems intended to detect bullying, whereas accuracy across all three classes can be misleading when neutral content dominates the distribution.</p>
</sec>
<sec id="sec16">
<label>6.2.3</label>
<title>Application of linguistic and psychological standards</title>
<p>Recent research has incorporated psychological theories to enhance cyberbullying detection by analyzing underlying personality traits in text (e.g., <xref ref-type="bibr" rid="ref32">Elzayady et al., 2023</xref>). Such frameworks align detection methods with broader behavioral standards, moving toward a more human-centered approach in identifying abusive content. Other studies, such as <xref ref-type="bibr" rid="ref26">Boulouard et al. (2022)</xref>, address multilingual standards by analyzing Arabic text in translation and leveraging cross-linguistic BERT models, thus ensuring consistency in detecting cyberbullying characteristics across languages.</p>
</sec>
</sec>
<sec id="sec17">
<label>6.3</label>
<title>Research question 3</title>
<p>The third RQ was:</p>
<p>What future research directions in cyberbullying detection may be established based on the findings of the provided systematic review?</p>
<p>The following themes were developed to answer the third research question.</p>
<sec id="sec18">
<label>6.3.1</label>
<title>Expansion of dialect-specific datasets and multilingual analysis</title>
<p>Future research could focus on developing larger, dialect-specific datasets to address the significant linguistic diversity in Arabic. Datasets for Moroccan, Syrian, and Gulf dialects remain limited and would improve detection accuracy for specific regions (e.g., <xref ref-type="bibr" rid="ref48">Rachidi et al., 2023</xref>; <xref ref-type="bibr" rid="ref11">Ali and Kurdy, 2022</xref>). Studies also suggest expanding multilingual capabilities to improve cross-linguistic performance, with transformer models like BERT and mBERT showing potential for multilingual hate speech analysis (e.g., <xref ref-type="bibr" rid="ref14">Alrashidi et al., 2023</xref>; <xref ref-type="bibr" rid="ref53">Shannaq et al., 2022</xref>).</p>
<p>For limited-resource settings, few strategies with large language models can be grounded in complementary lines of evidence. First, in-context learning has been shown to deliver strong few-shot performance without gradient updates; GPT-3&#x2019;s original study established that scaling enables task-agnostic adaptation via a handful of exemplars embedded in the prompt, a result that has shaped subsequent methodology for low-data regimes (<xref ref-type="bibr" rid="ref27">Brown et al., 2020</xref>). Second, prompt-based and prompt-free fine-tuning methods consistently improve over na&#x00EF;ve fine-tuning when labeled data are scarce. Pattern-Exploiting Training and its generative extension reframe supervision as cloze-style patterns to amplify supervision from very small datasets, while LM-BFF automates prompt construction and demonstration selection to yield large gains across classification and regression tasks (<xref ref-type="bibr" rid="ref51">Schick and Sch&#x00FC;tze, 2020</xref>). Complementing these, SetFit avoids handcrafted prompts altogether by contrastively fine-tuning sentence-transformer encoders on a handful of pairs and then training a lightweight classifier on the induced embeddings, matching or surpassing larger fully fine-tuned models under strict few-shot budgets (<xref ref-type="bibr" rid="ref54">Tunstall et al., 2022</xref>). Moreover, parameter-efficient adaptation techniques such as LoRA reduce trainable parameters by orders of magnitude while preserving or improving downstream quality, which is particularly attractive when domain transfer must be achieved under tight compute and annotation constraints (<xref ref-type="bibr" rid="ref37">Hu et al., 2022</xref>). To mitigate the scarcity of human-written instructions, Self-Instruct bootstraps synthetic instruction&#x2013;input&#x2013;output triplets from the model itself and shows substantial gains over the base model, offering a practical path when labeled data are limited (<xref ref-type="bibr" rid="ref56">Wang et al., 2022</xref>). Evidence from multilingual and domain-specific studies indicates that these approaches translate beyond English benchmarks. Cross-lingual in-context learning studies report consistent benefits for genuinely low-resource languages and highlight alignment techniques that stabilize label semantics across languages, while evaluations in biomedical and clinical NLP show that instruction-tuned LLMs can perform competitively on few-shot entity recognition, QA, and relation extraction when carefully prompted (<xref ref-type="bibr" rid="ref28">Cahyawijaya et al., 2024</xref>).</p>
</sec>
<sec id="sec19">
<label>6.3.2</label>
<title>Enhanced deep learning models and feature engineering</title>
<p>Future research could involve advancing feature engineering, particularly through contextual embeddings, attention mechanisms, and personality inference models. These methods could enhance the interpretability of cyberbullying detection systems and better capture contextual aspects of offensive language (e.g., <xref ref-type="bibr" rid="ref43">Mohaouchane et al., 2019</xref>; <xref ref-type="bibr" rid="ref32">Elzayady et al., 2023</xref>). Additionally, hybrid models combining CNN, RNN, and BERT-based architectures have shown promise for handling complex language features, and future studies could explore further model fusion or ensemble methods for improved accuracy (e.g., <xref ref-type="bibr" rid="ref43">Mohaouchane et al., 2019</xref>; <xref ref-type="bibr" rid="ref19">Althobaiti, 2022</xref>).</p>
</sec>
<sec id="sec20">
<label>6.3.3</label>
<title>Ethical considerations and real-time detection systems</title>
<p>Ethical standards and privacy concerns will play a growing role in future cyberbullying detection research. Privacy-preserving algorithms, especially those that anonymize or filter sensitive information, can support ethical AI use on social media platforms (e.g., <xref ref-type="bibr" rid="ref47">Omar et al., 2021</xref>). Another area for future exploration is real-time cyberbullying detection systems that respond dynamically to harmful content as it is posted. While challenging, real-time models could be feasible with lightweight DL architectures tailored for social media monitoring (e.g., <xref ref-type="bibr" rid="ref20">Amer Hamzah and Dhannoon, 2023</xref>; <xref ref-type="bibr" rid="ref39">Kanan et al., 2021</xref>).</p>
<p>Ethical risks arise at each stage of dataset development and deployment for Arabic cyberbullying detection, beginning with data collection. The Instagram-based benchmark demonstrates the value of reporting annotation protocols and inter-annotator agreement alongside careful corpus descriptions; however, as with Twitter- and YouTube-based datasets, the presence of user mentions and cross-post threads can inadvertently expose targets and perpetrators if not aggressively sanitized (<xref ref-type="bibr" rid="ref7">Albayari and Abdallah, 2022</xref>; <xref ref-type="bibr" rid="ref9008">Haidar et al., 2019</xref>; <xref ref-type="bibr" rid="ref6">Alakrot et al., 2018</xref>; <xref ref-type="bibr" rid="ref9002">Alduailaj et al., 2023</xref>; <xref ref-type="bibr" rid="ref9001">Al-Ajlan and Ykhlef, 2018</xref>; <xref ref-type="bibr" rid="ref9005">Alrougi et al., 2024</xref>). Representativeness is a second, persistent ethical and scientific concern. Arabic social media is heterogeneous across dialects, platforms, and communities; yet several widely used datasets skew toward particular dialect clusters or platform norms, such as Egyptian or Gulf Twitter, pan-Arab YouTube comments, or Instagram captions from specific demographic groups (<xref ref-type="bibr" rid="ref9008">Haidar et al., 2019</xref>). Studies that publish clear guidelines, show label distributions, and report inter-annotator agreement support more accountable modeling than those that provide only aggregate scores (<xref ref-type="bibr" rid="ref7">Albayari and Abdallah, 2022</xref>). Curators should also protect annotator wellbeing through workload limits, content warnings, and access to support, and they should state these safeguards in their documentation. The evaluation protocol has ethical implications because metric choice shapes decision thresholds used in practice. Practical architectures therefore favor lightweight normalizers and dialect-aware tokenization before model inference, with privacy-preserving logging that stores only hashed text fingerprints or short-lived embeddings for auditing (<xref ref-type="bibr" rid="ref6">Alakrot et al., 2018</xref>). The more explicit dataset papers are about these elements, the less likely it is that downstream models will inadvertently encode representational harms or privacy leakage.</p>
</sec>
<sec id="sec21">
<label>6.3.4</label>
<title>Integration of psychological and social dimensions</title>
<p>Integrating psychological and social analysis within detection algorithms is emerging as an essential direction. Personality-based approaches could be particularly useful, helping identify users more likely to engage in or be affected by cyberbullying (e.g., <xref ref-type="bibr" rid="ref32">Elzayady et al., 2023</xref>).</p>
<p>Additionally, cross-disciplinary research involving psychology, sociology, and computational linguistics could establish standards for understanding the social dynamics underlying cyberbullying, offering insights beyond linguistic patterns (e.g., <xref ref-type="bibr" rid="ref47">Omar et al., 2021</xref>). <xref ref-type="table" rid="tab4">Table 4</xref> shows the summary of the themes related to each research question.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Summary of the themes related to each research question.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Research Question</th>
<th align="left" valign="top">Theme</th>
<th align="left" valign="top">Description</th>
<th align="left" valign="top">Sources</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" rowspan="3">RQ1: Current trends in cyberbullying detection for Arabic language and dialects</td>
<td align="left" valign="top">ML and DL Approaches</td>
<td align="left" valign="top">ML models (e.g., SVM, Na&#x00EF;ve Bayes) and DL models (e.g., CNN, BERT) are common for cyberbullying detection, with ensemble methods improving accuracy.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref35">Haidar et al. (2017)</xref>; <xref ref-type="bibr" rid="ref6">Alakrot et al. (2018)</xref>; <xref ref-type="bibr" rid="ref14">Alrashidi et al. (2023)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Sentiment Analysis and Lexicon-Based Methods</td>
<td align="left" valign="top">Sentiment analysis and lexicon-based approaches capture emotional tones and harmful language, essential for handling Arabic&#x2019;s diverse dialects.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref9">AlHarbi et al. (2019)</xref>; <xref ref-type="bibr" rid="ref33">Farid and El-Tazi (2020)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Handling Arabic Dialects and Complexity</td>
<td align="left" valign="top">Specialized datasets and models (e.g., AraBERT, multilingual BERT) address dialectal variability, enhancing model accuracy for Arabic.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref45">Mubarak and Darwish (2019)</xref>; <xref ref-type="bibr" rid="ref1">AbdelHamid et al. (2022)</xref>; <xref ref-type="bibr" rid="ref41">Khezzar et al. (2023)</xref></td>
</tr>
<tr>
<td align="left" valign="top" rowspan="4">RQ2: Standards used for detecting cyberbullying based on its characteristics</td>
<td align="left" valign="top">Development of Cyberbullying Datasets</td>
<td align="left" valign="top">Creation of Arabic-specific datasets that include dialectical variations and cyberbullying characteristics, though issues like imbalanced datasets (few cyberbullying instances) impact model performance.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref23">Bashir and Bouguessa (2021)</xref>; <xref ref-type="bibr" rid="ref40">Khairy et al. (2023)</xref>; <xref ref-type="bibr" rid="ref1">AbdelHamid et al. (2022)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Evaluation Standards and Metrics</td>
<td align="left" valign="top">Precision, recall, F1-score, and accuracy are commonly used metrics, supplemented by specialized metrics tailored to Arabic-language characteristics to ensure reliable detection performance.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref35">Haidar et al. (2017)</xref>; <xref ref-type="bibr" rid="ref5">Alakrot et al. (2021)</xref>; <xref ref-type="bibr" rid="ref26">Boulouard et al. (2022)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Linguistic and Psychological Standards</td>
<td align="left" valign="top">Integration of linguistic and psychological insights, such as personality inference, allows a deeper understanding of user behavior, helping to identify cyberbullying based on more human-centered behavioral traits.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref32">Elzayady et al. (2023)</xref>; <xref ref-type="bibr" rid="ref47">Omar et al. (2021)</xref>; <xref ref-type="bibr" rid="ref53">Shannaq et al. (2022)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Contextual and Cultural Considerations</td>
<td align="left" valign="top">Incorporation of cultural sensitivity, including the use of dialect-specific language features, emojis, and contextual sentiment, provides a more nuanced and culturally accurate detection of offensive language.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref9">AlHarbi et al. (2019)</xref>; <xref ref-type="bibr" rid="ref33">Farid and El-Tazi (2020)</xref>; <xref ref-type="bibr" rid="ref41">Khezzar et al. (2023)</xref></td>
</tr>
<tr>
<td align="left" valign="top" rowspan="4">RQ3: Future research directions for Arabic cyberbullying detection</td>
<td align="left" valign="top">Dialect-Specific Datasets and Multilingual Models</td>
<td align="left" valign="top">Expansion of dialect-specific datasets and multilingual models to enhance detection across Arabic dialects and improve cross-linguistic applicability.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref11">Ali and Kurdy (2022)</xref>; <xref ref-type="bibr" rid="ref48">Rachidi et al. (2023)</xref>; <xref ref-type="bibr" rid="ref53">Shannaq et al. (2022)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Advanced Feature Engineering and Hybrid Models</td>
<td align="left" valign="top">Development of hybrid models (e.g., CNN-LSTM-BERT) and advanced feature engineering, such as attention mechanisms and personality-based features, for richer context and improved detection accuracy.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref44">Mouheb et al. (2019)</xref>; <xref ref-type="bibr" rid="ref32">Elzayady et al. (2023)</xref>; <xref ref-type="bibr" rid="ref26">Boulouard et al. (2022)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Real-Time Detection and Privacy Considerations</td>
<td align="left" valign="top">Focus on real-time cyberbullying detection models for immediate response, with privacy-preserving techniques to ensure user data protection and ethical AI application.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref20">Amer Hamzah and Dhannoon (2023)</xref>; <xref ref-type="bibr" rid="ref47">Omar et al. (2021)</xref>; <xref ref-type="bibr" rid="ref39">Kanan et al. (2021)</xref></td>
</tr>
<tr>
<td align="left" valign="top">Cross-Disciplinary Research</td>
<td align="left" valign="top">Integration of psychological, sociological, and linguistic insights for a more comprehensive understanding of the social and behavioral dynamics underlying Arabic cyberbullying.</td>
<td align="left" valign="top"><xref ref-type="bibr" rid="ref33">Farid and El-Tazi (2020)</xref>; <xref ref-type="bibr" rid="ref47">Omar et al. (2021)</xref>; <xref ref-type="bibr" rid="ref32">Elzayady et al. (2023)</xref></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The results of the research emphasize the necessity of culturally sensitive detection models, sophisticated methodologies, and tailored approaches to effectively capture the distinctive characteristics of the Arabic offensive language. Arabic is an extremely diverse language, with significant variations in dialects across regions (e.g., Egyptian, Gulf, Levantine), each with its own vocabulary, syntax, and expressions. The detection of objectionable language is further complicated by this diversity, as models that have been trained in Modern Standard Arabic frequently encounter difficulties with dialectal content. These results suggest that the model&#x2019;s ability to identify nuanced or implicit forms of offensive language, such as sarcasm or mockery, is improved by the inclusion of sentiment and lexicon-based features that are specifically designed for Arabic dialects and slang. Many categories of offensive language, including religious hate speech, ethnic hate, and political offence, have been classified by researchers. These types of language are particularly sensitive in Arabic-speaking societies. These categories are indicative of regional and cultural priorities, emphasizing the social and religious values that influence online discourse in Arabic contexts. The importance of accounting for these categories is underscored by research, as they pertain to highly sensitive subjects that may vary in severity and context in comparison to other languages. The results indicate that culturally aware models that identify these particular forms of objectionable language can improve the accuracy and relevance of the models.</p>
<p>Although numerous studies have examined cyberbullying detection methods broadly or across various languages, there is a paucity of focused analyses on Arabic-language detection, given the unique challenges presented by Arabic&#x2019;s morphological intricacies and dialectal diversity (<xref ref-type="bibr" rid="ref45">Mubarak and Darwish, 2019</xref>; <xref ref-type="bibr" rid="ref1">AbdelHamid et al., 2022</xref>). The majority of the earlier studies predominantly analyze general patterns in cyberbullying detection, concentrating on English-language research (<xref ref-type="bibr" rid="ref6">Alakrot et al., 2018</xref>; <xref ref-type="bibr" rid="ref23">Bashir and Bouguessa, 2021</xref>). Although current studies recognize dataset imbalances and biases in social media-derived training data, they frequently neglect to consider privacy concerns and the ethical ramifications of automated cyberbullying detection among Arabic-speaking groups (<xref ref-type="bibr" rid="ref47">Omar et al., 2021</xref>; <xref ref-type="bibr" rid="ref20">Amer Hamzah and Dhannoon, 2023</xref>). This study addresses real-time detection concerns, the balance between moderation and free speech, and the necessity for privacy-preserving machine learning algorithms in social media monitoring (<xref ref-type="bibr" rid="ref39">Kanan et al., 2021</xref>). This paper distinctly focuses on the thorough assessment of ML and DL models in detecting cyberbullying in Arabic. The prior systematic literature review by <xref ref-type="bibr" rid="ref29">Casta&#x00F1;o-Pulgar&#x00ED;n et al. (2021)</xref>, addressed cyberbullying detection on studies that provided exploratory data about the Internet and social media as a space for online hate speech, types of cyberhate, terrorism as an online hate trigger, online hate expressions and the most common methods to assess online hate speech. <xref ref-type="bibr" rid="ref22">Balakrisnan and Kaity (2023)</xref> also did an SLR focusing on three main areas regarding cyberbullying detection through machine learning: the algorithms employed, the features used for detection, and the performance measures of these methods. The prior studies and reviews neglect Arabic-specific issues such as root-based word creation, tokenization complexities, and script intricacies.</p>
<p>The results of this study underscore the necessity of creating extensive, dialect-specific datasets and enhancing NLP models to address syntactic and lexical discrepancies among Arabic dialects. Deep learning architectures such as CNNs and BiLSTMs generally surpass classical baselines once training sets exceed the low-thousands and when preprocessed to handle orthographic variation, elongation, and code-mixing. Transformer models fine-tuned on Arabic corpora&#x2014;especially variants trained with substantial dialectal coverage&#x2014;consistently lead when the label definitions align with the pretraining distribution and when macro-averaged F1 rather than accuracy guides optimization. A recurring empirical pattern is precision outpacing recall, reflecting systems that confidently flag explicit bullying but struggle with implicit attacks, sarcasm, and context-dependent harassment. Performance differences are driven first by data composition. Dialectal diversity, platform genre, and class design are the most decisive factors. Models trained on tweets in Egyptian or Gulf dialects tend to degrade on Levantine, Maghrebi, or code-mixed content because lexical cues and morphological patterns shift, and subword tokenizers learned on Modern Standard Arabic under-segment dialectal forms. Domain shift between platforms&#x2014;short, slang-heavy tweets versus longer Instagram captions or YouTube comments&#x2014;likewise reduces transfer, as does the prevalence of emojis, creative spellings, and Arabizi. Class definitions also vary: some corpora equate cyberbullying with general abuse or toxicity, whereas others require intent, repetition, or power imbalance. The broader the &#x201C;bullying&#x201D; label, the higher the apparent scores, but the weaker the comparability across studies. Evaluation choices amplify these effects. Where annotation guidelines were explicit and inter-annotator agreement documented, models learned more stable decision boundaries; where guidelines were minimal or borrowed from sentiment analysis, models overfit to superficial polarity and miss community-specific bullying norms. Pretraining and representation learning explain the remaining variance. Yet, when fine-tuning data are severely imbalanced, even strong encoders prioritize surface toxicity over nuanced bullying constructs. In contrast, classical models augmented with curated lexicons and character-level features sometimes outperform deep baselines on noisy, low-resource dialects because they are less sensitive to tokenization errors and require fewer examples to generalize.</p>
<p>The most promising methodological direction is dialect- and domain-robust modeling anchored in standardized evaluation. Progress depends on a benchmark suite that harmonizes label schemas for cyberbullying versus general abuse, publishes class priors, and mandates macro-F1 and per-class F1 with clear treatment of the neutral class. Cross-dataset testing should be routine, with models trained on one corpus evaluated zero-shot on another to measure real-world robustness. Data and supervision strategies also offer leverage. Active learning and disagreement-focused annotation can densify minority bullying phenomena such as threats, doxxing, or body-shaming. Weak supervision that combines lexicon rules, community guidelines, and pattern matchers can cheaply label large pools for pretraining, followed by human verification on hard examples. Span-level rationales and multi-label tags for bullying types improve transparency and enable error analysis beyond single-label outcomes, while adversarial training with paraphrases and sarcasm transformations hardens models against implicit aggression. Context modeling is a further frontier. Many failures stem from sentence-level isolation. Incorporating conversation threads, author&#x2013;target history, and lightweight social signals can disambiguate teasing from harassment and detect repetition, a hallmark of bullying. Graph-based representations of interactions, when coupled with privacy-preserving design and strict ethical safeguards, can capture power asymmetries and coordinated attacks without storing sensitive personal attributes.</p>
<p>Finally, instruction-tuned large language models adapted to Arabic show potential as few-shot labelers, error analyzers, and data generators, but their deployment must be paired with rigorous calibration, bias auditing across dialects and demographics, and conservative thresholding in safety-critical pipelines. Taken together, the evidence suggests that the field is moving from accuracy on single, homogeneous datasets toward robust, dialect-inclusive systems evaluated under standardized, recall-sensitive protocols, with the integration of context and improved supervision likely to yield the next substantive gains.</p>
</sec>
</sec>
</sec>
<sec id="sec22">
<label>7</label>
<title>Limitations and suggestions for future studies</title>
<p>A key limitation of this review is the absence of a formal quality appraisal or risk-of-bias assessment of the included studies. Established tools such as AMSTAR, AMSTAR-2, or ROBIS are often used in systematic reviews to evaluate the methodological rigor of primary studies and to distinguish between stronger and weaker evidence. The present review treats all included studies as methodologically equivalent, regardless of variations in their design, sampling strategies, or analytical robustness.</p>
<p>The majority of the studies reviewed are based on restricted or specific datasets, which may not adequately represent the complete range of Arabic dialectal diversity or the diverse forms of cyberbullying that are present on different platforms. However, the absence of standardized datasets for the detection of Arabic cyberbullying also presents obstacles to the attainment of generalizable results. Despite the potential of dialect-specific models, the complexity and extensive variations among Arabic dialects pose a significant obstacle. The results may not be broadly applicable because current models may not perform consistently across all dialects. The detection of real-time cyberbullying is still in its infancy, particularly in the context of Arabic texts. Although some studies incorporate psychological insights, there is a void in the comprehensive integration of insights from sociology, linguistics, and psychology to develop a holistic understanding of cyberbullying behaviors specific to Arabic-speaking regions. Another limitation of this review is the exclusion of conference proceedings, despite their prominence as venues for innovation in natural language processing. Nonetheless, this exclusion may have led to the omission of some cutting-edge contributions. Future reviews should consider incorporating both journal articles and high-quality conference proceedings to present a more comprehensive view of the research landscape.</p>
<p>Future research may investigate sophisticated deep learning architectures and hybrid models that amalgamate various methodologies to enhance detection, to improve contextual comprehension and classification precision. Another vital avenue for future study is the enhancement of sentiment-based and context-aware models for detecting cyberbullying. The problem of dataset imbalance persists, as cases of cyberbullying are markedly underrepresented relative to non-offensive content.</p>
</sec>
<sec sec-type="conclusions" id="sec23">
<label>8</label>
<title>Conclusion</title>
<p>This study offers a thorough examination of the most recent academic research, methodologies, and challenges in the detection of cyberbullying in Arabic texts. This review emphasizes the substantial advancements that have been achieved in this field by evaluating the efficacy of ML and DL models, sentiment analysis, lexicon-based methods, and dialectal considerations. The significance of specialized datasets for Arabic dialects, the efficacy of composite models and ensemble learning, and the value of sentiment-based and contextual analysis are underscored by the key findings.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec24">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author/s.</p>
</sec>
<sec sec-type="author-contributions" id="sec25">
<title>Author contributions</title>
<p>HA: Conceptualization, Data curation, Investigation, Methodology, Project administration, Software, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. MA: Investigation, Methodology, Project administration, Supervision, Validation, Writing &#x2013; review &#x0026; editing. SM: Methodology, Project administration, Supervision, Validation, Writing &#x2013; review &#x0026; editing. AB: Project administration, Software, Supervision, Validation, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec sec-type="funding-information" id="sec26">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. The authors extend their appreciation to the Deanship of Scientific Research at Northern Border University, Arar, KSA for funding this research work through the project number &#x201C;NBU-SAFIR-2025&#x201D;.</p>
</sec>
<ack>
<p>The authors would like to thank their academic peers and institutional colleagues for their feedback during the early stages of this research.</p>
</ack>
<sec sec-type="COI-statement" id="sec27">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec28">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec29">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>AbdelHamid</surname><given-names>M.</given-names></name> <name><surname>Jafar</surname><given-names>A.</given-names></name> <name><surname>Rahal</surname><given-names>Y.</given-names></name></person-group> (<year>2022</year>). <article-title>Levantine hate speech detection in twitter</article-title>. <source>Soc. Netw. Anal. Min.</source> <volume>12</volume>:<fpage>121</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s13278-022-00950-4</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abdelmonem</surname><given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Reconceptualizing sexual harassment in Egypt: a longitudinal assessment of el-Taharrush el-Ginsy in Arabic online forums and anti-sexual harassment activism</article-title>. <source>Kohl: J. Body Gender Res.</source> <volume>1</volume>, <fpage>23</fpage>&#x2013;<lpage>41</lpage>. doi: <pub-id pub-id-type="doi">10.36583/kohl/1-1/</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Abu Farha</surname><given-names>I.</given-names></name></person-group> (<year>2023</year>). Arabic sarcasm detection.</citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al</surname><given-names>Z. N.</given-names></name></person-group> (<year>2019</year>). <article-title>Divine impoliteness: how Arabs negotiate Islamic moral order on twitter</article-title>. <source>Russ. J. Linguist.</source> <volume>23</volume>, <fpage>1039</fpage>&#x2013;<lpage>1064</lpage>.</citation></ref>
<ref id="ref5"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Alakrot</surname><given-names>A.</given-names></name> <name><surname>Fraifer</surname><given-names>M.</given-names></name> <name><surname>Nikolov</surname><given-names>N. S.</given-names></name></person-group> (<year>2021</year>). &#x201C;Machine learning approach to detection of offensive language in online communication in Arabic.&#x201D; in <italic>2021 IEEE 1st international Maghreb meeting of the conference on sciences and techniques of automatic control and computer engineering MI-STA</italic>, pp. 244&#x2013;249.</citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alakrot</surname><given-names>A.</given-names></name> <name><surname>Murray</surname><given-names>L.</given-names></name> <name><surname>Nikolov</surname><given-names>N. S.</given-names></name></person-group> (<year>2018</year>). <article-title>Towards accurate detection of offensive language in online communication in Arabic</article-title>. <source>Proc. Comput. Sci.</source> <volume>142</volume>, <fpage>315</fpage>&#x2013;<lpage>320</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.procs.2018.10.491</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Albayari</surname><given-names>R.</given-names></name> <name><surname>Abdallah</surname><given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>Instagram-based benchmark dataset for cyberbullying detection in Arabic text</article-title>. <source>Data</source> <volume>7</volume>:<fpage>83</fpage>. doi: <pub-id pub-id-type="doi">10.3390/data7070083</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="book"><person-group person-group-type="author"><name><surname>AlFarah</surname><given-names>M. E.</given-names></name> <name><surname>Kamel</surname><given-names>I.</given-names></name> <name><surname>Al Aghbari</surname><given-names>Z.</given-names></name> <name><surname>Mouheb</surname><given-names>D.</given-names></name></person-group> (<year>2022</year>). &#x201C;<article-title>Arabic cyberbullying detection from imbalanced dataset using machine learning</article-title>&#x201D; in <source>Soft computing and its engineering applications</source>. eds. <person-group person-group-type="editor"><name><surname>Patel</surname><given-names>K. K.</given-names></name> <name><surname>Doctor</surname><given-names>G.</given-names></name> <name><surname>Patel</surname><given-names>A.</given-names></name> <name><surname>Lingras</surname><given-names>P.</given-names></name></person-group>, vol. <volume>1572</volume> (Changa, Anand, India: <publisher-name>Springer International Publishing</publisher-name>), <fpage>397</fpage>&#x2013;<lpage>409</lpage>. doi: <pub-id pub-id-type="doi">10.1007/978-3-031-05767-0_31</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="other"><person-group person-group-type="author"><name><surname>AlHarbi</surname><given-names>B. Y.</given-names></name> <name><surname>AlHarbi</surname><given-names>M. S.</given-names></name> <name><surname>AlZahrani</surname><given-names>N. J.</given-names></name> <name><surname>Alsheail</surname><given-names>M. M.</given-names></name> <name><surname>Alshobaili</surname><given-names>J. F.</given-names></name> <name><surname>Ibrahim</surname><given-names>D. M.</given-names></name></person-group> (<year>2019</year>). <italic>Automatic cyber bullying detection in Arabic social media</italic>. 12(12).</citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al-Hassan</surname><given-names>A.</given-names></name> <name><surname>Al-Dossari</surname><given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>Detection of hate speech in Arabic tweets using deep learning</article-title>. <source>Multimedia Systems</source> <volume>28</volume>, <fpage>1963</fpage>&#x2013;<lpage>1974</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00530-020-00742-w</pub-id></citation></ref>
<ref id="ref9001"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Al-Ajlan</surname><given-names>M. A.</given-names></name> <name><surname>Ykhlef</surname><given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>Optimized Twitter cyberbullying detection based on deep learning</article-title>. In <source>Proceedings of the 2018 21st Saudi Computer Society National Computer Conference (NCC)</source>. <fpage>1</fpage>&#x2013;<lpage>5</lpage>. <publisher-name>IEEE</publisher-name>. doi: <pub-id pub-id-type="doi">10.1109/NCG.2018.8593146</pub-id></citation></ref>
<ref id="ref9002"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alduailaj</surname><given-names>A. M.</given-names></name> <name><surname>Belghith</surname><given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Detecting Arabic cyberbullying tweets using machine learning</article-title>. <source>Mach. Learn. Knowl. Extr.</source> <volume>5</volume>, <fpage>29</fpage>&#x2013;<lpage>42</lpage>. doi: <pub-id pub-id-type="doi">10.3390/make5010003</pub-id></citation></ref>
<ref id="ref9003"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alhashmi</surname><given-names>A. A.</given-names></name> <name><surname>Darem</surname><given-names>A. A.</given-names></name></person-group> (<year>2022</year>). <article-title>Consensus-based ensemble model for Arabic cyberbullying detection</article-title>. <source>Computer Systems Science and Engineering</source>, <volume>41</volume>, <fpage>241</fpage>&#x2013;<lpage>254</lpage>. doi: <pub-id pub-id-type="doi">10.32604/csse.2022.020023</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ali</surname><given-names>R.</given-names></name> <name><surname>Kurdy</surname><given-names>D. M.-B.</given-names></name></person-group> (<year>2022</year>). <article-title>Cyberbullying detection in Syrian slang on social media by using data mining</article-title>. <source>Int. J. Eng. Res.</source> <volume>11</volume>.</citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aljalaoud</surname><given-names>H.</given-names></name> <name><surname>Dashtipour</surname><given-names>K.</given-names></name> <name><surname>AI Dubai</surname><given-names>A.</given-names></name></person-group> (<year>2025</year>). <article-title>Arabic cyberbullying detection: a comprehensive review of datasets and methodologies</article-title>. <source>IEEE Access</source>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2025.3561132</pub-id></citation></ref>
<ref id="ref9004"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aljarah</surname><given-names>I.</given-names></name> <name><surname>Habib</surname><given-names>M.</given-names></name> <name><surname>Hijazi</surname><given-names>N.</given-names></name> <name><surname>Faris</surname><given-names>H.</given-names></name> <name><surname>Qaddoura</surname><given-names>R.</given-names></name> <name><surname>Hammo</surname><given-names>B.</given-names></name> <etal/></person-group> (<year>2021</year>). <article-title>Intelligent detection of hate speech in Arabic social network: A machine learning approach</article-title>. <source>J. Inf. Sci.</source> <volume>47</volume>, <fpage>483</fpage>&#x2013;<lpage>501</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0165551520917651</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aljuhani</surname><given-names>O.</given-names></name> <name><surname>Alyoubi</surname><given-names>K.</given-names></name> <name><surname>Alotaibi</surname><given-names>F.</given-names></name></person-group> (<year>2022</year>). <article-title>Detecting Arabic offensive language in microblogs using domain-specific word Embeddings and deep learning</article-title>. <source>Tehni&#x010D;ki Glasnik</source> <volume>16</volume>, <fpage>394</fpage>&#x2013;<lpage>400</lpage>. doi: <pub-id pub-id-type="doi">10.31803/tg-20220305120018</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alrashidi</surname><given-names>B.</given-names></name> <name><surname>Jamal</surname><given-names>A.</given-names></name> <name><surname>Alkhathlan</surname><given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>Abusive content detection in Arabic tweets using multi-task learning and transformer-based models</article-title>. <source>Appl. Sci.</source> <volume>13</volume>:<fpage>5825</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app13105825</pub-id></citation></ref>
<ref id="ref9005"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alrougi</surname><given-names>M.</given-names></name> <name><surname>Alamoudi</surname><given-names>G.</given-names></name> <name><surname>Algamdi</surname><given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>ArbCyD: An Arabic post dataset for cyberbullying detection</article-title>. <source>J. Electr. Syst.</source> <volume>20</volume>, <fpage>1583</fpage>&#x2013;<lpage>1589</lpage>.</citation></ref>
<ref id="ref15"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Alsafari</surname><given-names>S.</given-names></name> <name><surname>Sadaoui</surname><given-names>S.</given-names></name> <name><surname>Mouhoub</surname><given-names>M.</given-names></name></person-group> (<year>2020a</year>). &#x201C;Deep Learning Ensembles for Hate Speech Detection.&#x201D; in <italic>2020 IEEE 32nd International Conference on Tools with Artificial Intelligence (ICTAI)</italic>, pp. 526&#x2013;531.</citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alsafari</surname><given-names>S.</given-names></name> <name><surname>Sadaoui</surname><given-names>S.</given-names></name> <name><surname>Mouhoub</surname><given-names>M.</given-names></name></person-group> (<year>2020b</year>). <article-title>Hate and offensive speech detection on Arabic social media</article-title>. <source>Online Soc. Networks Media</source> <volume>19</volume>:<fpage>100096</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.osnem.2020.10009</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alshalabi</surname><given-names>N.</given-names></name> <name><surname>Lahiani</surname><given-names>H.</given-names></name> <name><surname>Yasin</surname><given-names>A.</given-names></name></person-group> (<year>2024</year>). <article-title>The role of culture in abusive language on social media: examining the use of English and Arabic derogatory terms</article-title>. <source>Theory Pract. Lang. Stud.</source> <volume>14</volume>, <fpage>3057</fpage>&#x2013;<lpage>3066</lpage>. doi: <pub-id pub-id-type="doi">10.17507/tpls.1410.06</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alsubait</surname><given-names>T.</given-names></name> <name><surname>Alfageh</surname><given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>Comparison of machine learning techniques for cyberbullying detection on YouTube Arabic comments</article-title>. <source>Int. J. Comput. Sci. Netw. Secur.</source> <volume>21</volume>, <fpage>1</fpage>&#x2013;<lpage>5</lpage>. doi: <pub-id pub-id-type="doi">10.22937/IJCSNS.2021.21.1.1</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Althobaiti</surname><given-names>M. J.</given-names></name></person-group> (<year>2022</year>). <article-title>BERT-based approach to Arabic hate speech and offensive language detection in twitter: exploiting emojis and sentiment analysis</article-title>. <source>Int. J. Adv. Comput. Sci. Appl.</source> <volume>13</volume>:<fpage>5109</fpage>. doi: <pub-id pub-id-type="doi">10.14569/IJACSA.2022.01305109</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Amer Hamzah</surname><given-names>N.</given-names></name> <name><surname>Dhannoon</surname><given-names>B. N.</given-names></name></person-group> (<year>2023</year>). <article-title>Detecting Arabic sexual harassment using bidirectional long-short-term memory and a temporal convolutional network</article-title>. <source>Egypt. Inform. J.</source> <volume>24</volume>, <fpage>365</fpage>&#x2013;<lpage>373</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.eij.2023.05.007</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Anezi</surname><given-names>F. Y. A.</given-names></name></person-group> (<year>2022</year>). <article-title>Arabic hate speech detection using deep recurrent neural networks</article-title>. <source>Appl. Sci.</source> <volume>12</volume>:<fpage>6010</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app12126010</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Balakrisnan</surname><given-names>V.</given-names></name> <name><surname>Kaity</surname><given-names>M.</given-names></name></person-group> (<year>2023</year>). <article-title>Cyberbullying detection and machine learning: a systematic literature review</article-title>. <source>Artif. Intell. Rev.</source> <volume>56</volume>, <fpage>1375</fpage>&#x2013;<lpage>1416</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10462-023-10553-w</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bashir</surname><given-names>E.</given-names></name> <name><surname>Bouguessa</surname><given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Data mining for cyberbullying and harassment detection in Arabic texts</article-title>. <source>Int. J. Inform. Technol. Comp. Sci.</source> <volume>13</volume>, <fpage>41</fpage>&#x2013;<lpage>50</lpage>. doi: <pub-id pub-id-type="doi">10.5815/ijitcs.2021.05.04</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bertini</surname><given-names>F.</given-names></name> <name><surname>Allevi</surname><given-names>D.</given-names></name> <name><surname>Lutero</surname><given-names>G.</given-names></name> <name><surname>Montesi</surname><given-names>D.</given-names></name> <name><surname>Calz&#x00E0;</surname><given-names>L.</given-names></name></person-group> (<year>2021</year>). <article-title>Automatic speech classifier for mild cognitive impairment and early dementia</article-title>. <source>ACM Trans. Comp. Healthcare</source> <volume>3</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3469089</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Bouhlila</surname><given-names>D. S.</given-names></name></person-group> (<year>2019</year>). Sexual harassment and domestic violence in the Middle East and North Africa. Arab Barometer, 2.</citation></ref>
<ref id="ref9006"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bouliche</surname><given-names>A.</given-names></name> <name><surname>Rezoug</surname><given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>Detection of cyberbullying in Arabic social media using dynamic graph neural network</article-title>. In <source>Proceedings of the 1st Tunisian-Algerian Joint Conference on Applied Computing (TACC 2022)</source>. <fpage>1</fpage>&#x2013;<lpage>11</lpage>.</citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boulouard</surname><given-names>Z.</given-names></name> <name><surname>Ouaissa</surname><given-names>M.</given-names></name> <name><surname>Ouaissa</surname><given-names>M.</given-names></name> <name><surname>Krichen</surname><given-names>M.</given-names></name> <name><surname>Almutiq</surname><given-names>M.</given-names></name> <name><surname>Gasmi</surname><given-names>K.</given-names></name></person-group> (<year>2022</year>). <article-title>Detecting hateful and offensive speech in Arabic social media using transfer learning</article-title>. <source>Appl. Sci.</source> <volume>12</volume>:<fpage>12823</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app122412823</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brown</surname><given-names>T.</given-names></name> <name><surname>Mann</surname><given-names>B.</given-names></name> <name><surname>Ryder</surname><given-names>N.</given-names></name> <name><surname>Subbiah</surname><given-names>M.</given-names></name> <name><surname>Kaplan</surname><given-names>J. D.</given-names></name> <name><surname>Dhariwal</surname><given-names>P.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Language models are few-shot learners</article-title>. <source>Adv. Neural Inf. Proces. Syst.</source> <volume>33</volume>, <fpage>1877</fpage>&#x2013;<lpage>1901</lpage>.</citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cahyawijaya</surname><given-names>S.</given-names></name> <name><surname>Lovenia</surname><given-names>H.</given-names></name> <name><surname>Fung</surname><given-names>P.</given-names></name></person-group> (<year>2024</year>). <article-title>Llms are few-shot in-context low-resource language learners</article-title>. <source>arXiv preprint arXiv</source>:<fpage>2403.16512</fpage>.</citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Casta&#x00F1;o-Pulgar&#x00ED;n</surname><given-names>S. A.</given-names></name> <name><surname>Su&#x00E1;rez-Betancur</surname><given-names>N.</given-names></name> <name><surname>Vega</surname><given-names>L. M. T.</given-names></name> <name><surname>L&#x00F3;pez</surname><given-names>H. M. H.</given-names></name></person-group> (<year>2021</year>). <article-title>Internet, social media and online hate speech. Systematic review</article-title>. <source>Aggress. Violent Behav.</source> <volume>58</volume>:<fpage>101608</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.avb.2021.101608</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cowie</surname><given-names>J.</given-names></name> <name><surname>Lehnert</surname><given-names>W.</given-names></name></person-group> (<year>1996</year>). <article-title>Information extraction</article-title>. <source>Commun. ACM</source> <volume>39</volume>, <fpage>80</fpage>&#x2013;<lpage>91</lpage>. doi: <pub-id pub-id-type="doi">10.1145/234173.234209</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>El-Alami</surname><given-names>F.</given-names></name> <name><surname>Ouatik El Alaoui</surname><given-names>S.</given-names></name> <name><surname>En Nahnahi</surname><given-names>N.</given-names></name></person-group> (<year>2022</year>). <article-title>A multilingual offensive language detection method based on transfer learning from transformer fine-tuning model</article-title>. <source>J. King Saud Univ. - Comput. Inf. Sci.</source> <volume>34</volume>, <fpage>6048</fpage>&#x2013;<lpage>6056</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jksuci.2021.07.013</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elzayady</surname><given-names>H.</given-names></name> <name><surname>Mohamed</surname><given-names>M. S.</given-names></name> <name><surname>Badran</surname><given-names>K. M.</given-names></name> <name><surname>Salama</surname><given-names>G. I.</given-names></name></person-group> (<year>2023</year>). <article-title>A hybrid approach based on personality traits for hate speech detection in Arabic social media</article-title>. <source>Int. J. Elect. Comp. Eng.</source> <volume>13</volume>:<fpage>1979</fpage>&#x2013;88. doi: <pub-id pub-id-type="doi">10.11591/ijece.v13i2.pp1979-1988</pub-id></citation></ref>
<ref id="ref33"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Farid</surname><given-names>D.</given-names></name> <name><surname>El-Tazi</surname><given-names>D. N.</given-names></name></person-group> (<year>2020</year>). Detection of cyberbullying in tweets in Egyptian dialects. 18(7).</citation></ref>
<ref id="ref9007"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fati</surname><given-names>S. M.</given-names></name></person-group> (<year>2022</year>). <article-title>Detecting cyberbullying across social media platforms in Saudi Arabia using sentiment analysis: A case study</article-title>. <source>Comput. J</source>. <volume>65</volume>, <fpage>1787</fpage>&#x2013;<lpage>1794</lpage>. doi: <pub-id pub-id-type="doi">10.1093/comjnl/bxab019</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gr&#x00E9;goire</surname><given-names>Y.</given-names></name> <name><surname>Salle</surname><given-names>A.</given-names></name> <name><surname>Tripp</surname><given-names>T. M.</given-names></name></person-group> (<year>2015</year>). <article-title>Managing social media crises with your customers: the good, the bad, and the ugly</article-title>. <source>Bus. Horiz.</source> <volume>58</volume>, <fpage>173</fpage>&#x2013;<lpage>182</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bushor.2014.11.001</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haidar</surname><given-names>B.</given-names></name> <name><surname>Chamoun</surname><given-names>M.</given-names></name> <name><surname>Serhrouchni</surname><given-names>A.</given-names></name></person-group> (<year>2017</year>). <article-title>A multilingual system for cyberbullying detection: Arabic content detection using machine learning</article-title>. <source>Adv. Sci. Technol. Eng. Syst. J.</source> <volume>2</volume>, <fpage>275</fpage>&#x2013;<lpage>284</lpage>. doi: <pub-id pub-id-type="doi">10.25046/aj020634</pub-id></citation></ref>
<ref id="ref9008"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Haidar</surname><given-names>B.</given-names></name> <name><surname>Chamoun</surname><given-names>M.</given-names></name> <name><surname>Serhrouchni</surname><given-names>A.</given-names></name></person-group> (<year>2019</year>). <article-title>Arabic cyberbullying detection: Enhancing performance by using ensemble machine learning</article-title>. In <source>2019 International Conference on Internet of Things (iThings) and IEEE Green Computing and Communications (GreenCom) and IEEE Cyber, Physical and Social Computing (CPSCom) and IEEE Smart Data (SmartData)</source>. <fpage>323</fpage>&#x2013;<lpage>327</lpage>. <publisher-name>IEEE</publisher-name>. doi: <pub-id pub-id-type="doi">10.1109/iThings/GreenCom/CPSCom/SmartData.2019.00074</pub-id></citation></ref>
<ref id="ref36"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Haidar</surname><given-names>B.</given-names></name> <name><surname>Chamoun</surname><given-names>M.</given-names></name> <name><surname>Serhrouchni</surname><given-names>A</given-names></name></person-group> (<year>2018</year>). &#x201C;Arabic cyberbullying detection: Using deep learning.&#x201D; in 7th International Conference on Computer and Communication Engineering (ICCCE), IEEE. pp. 284&#x2013;289.</citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname><given-names>E. J.</given-names></name> <name><surname>Shen</surname><given-names>Y.</given-names></name> <name><surname>Wallis</surname><given-names>P.</given-names></name> <name><surname>Allen-Zhu</surname><given-names>Z.</given-names></name> <name><surname>Li</surname><given-names>Y.</given-names></name> <name><surname>Wang</surname><given-names>S.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Lora: low-rank adaptation of large language models</article-title>. <source>ICLR</source> <volume>1</volume>:<fpage>3</fpage>.</citation></ref>
<ref id="ref38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kanan</surname><given-names>T.</given-names></name> <name><surname>Aldaaja</surname><given-names>A.</given-names></name> <name><surname>Hawashin</surname><given-names>B.</given-names></name></person-group> (<year>2020</year>). <article-title>Cyber-bullying and cyber-harassment detection using supervised machine learning techniques in Arabic social media contents</article-title>. <source>J. Internet Technol.</source> <volume>21</volume>, <fpage>1409</fpage>&#x2013;<lpage>1421</lpage>.</citation></ref>
<ref id="ref39"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Kanan</surname><given-names>T.</given-names></name> <name><surname>Kanaan</surname><given-names>G. G.</given-names></name> <name><surname>Al-Shalabi</surname><given-names>R.</given-names></name> <name><surname>Aldaaja</surname><given-names>A.</given-names></name></person-group> (<year>2021</year>). Offensive language detection in social networks for Arabic language using clustering techniques.</citation></ref>
<ref id="ref40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khairy</surname><given-names>M.</given-names></name> <name><surname>Mahmoud</surname><given-names>T. M.</given-names></name> <name><surname>Omar</surname><given-names>A.</given-names></name> <name><surname>Abd El-Hafeez</surname><given-names>T.</given-names></name></person-group> (<year>2023</year>). <article-title>Comparative performance of ensemble machine learning for Arabic cyberbullying and offensive language detection</article-title>. <source>Lang. Resour. Eval.</source> <volume>58</volume>, <fpage>695</fpage>&#x2013;<lpage>712</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10579-023-09683-y</pub-id></citation></ref>
<ref id="ref41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khezzar</surname><given-names>R.</given-names></name> <name><surname>Moursi</surname><given-names>A.</given-names></name> <name><surname>Al Aghbari</surname><given-names>Z.</given-names></name></person-group> (<year>2023</year>). <article-title>ArHatedetector: detection of hate speech from standard and dialectal Arabic tweets</article-title>. <source>Discov. Internet Things</source> <volume>3</volume>:<fpage>1</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s43926-023-00030-9</pub-id></citation></ref>
<ref id="ref42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Miro&#x0144;czuk</surname><given-names>M. M.</given-names></name> <name><surname>Protasiewicz</surname><given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>A recent overview of the state-of-the-art elements of text classification</article-title>. <source>Expert Syst. Appl.</source> <volume>106</volume>, <fpage>36</fpage>&#x2013;<lpage>54</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.eswa.2018.03.058</pub-id></citation></ref>
<ref id="ref43"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Mohaouchane</surname><given-names>H.</given-names></name> <name><surname>Mourhir</surname><given-names>A.</given-names></name> <name><surname>Nikolov</surname><given-names>N. S.</given-names></name></person-group> (<year>2019</year>). &#x201C;Detecting Offensive Language on Arabic Social Media Using Deep Learning.&#x201D; in <italic>2019 Sixth International Conference on Social Networks Analysis, Management and Security (SNAMS),</italic> pp. 466&#x2013;471.</citation></ref>
<ref id="ref44"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Mouheb</surname><given-names>D.</given-names></name> <name><surname>Albarghash</surname><given-names>R.</given-names></name> <name><surname>Mowakeh</surname><given-names>M. F.</given-names></name> <name><surname>Aghbari</surname><given-names>Z. A.</given-names></name> <name><surname>Kamel</surname><given-names>I.</given-names></name></person-group> (<year>2019</year>). Detection of Arabic cyberbullying on social networks using machine learning.</citation></ref>
<ref id="ref45"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Mubarak</surname><given-names>H.</given-names></name> <name><surname>Darwish</surname><given-names>K.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>Arabic offensive language classification on twitter</article-title>&#x201D; in <source>Social informatics</source>. eds. <person-group person-group-type="editor"><name><surname>Weber</surname><given-names>I.</given-names></name> <name><surname>Darwish</surname><given-names>K. M.</given-names></name> <name><surname>Wagner</surname><given-names>C.</given-names></name> <name><surname>Zagheni</surname><given-names>E.</given-names></name> <name><surname>Nelson</surname><given-names>L.</given-names></name> <name><surname>Aref</surname><given-names>S.</given-names></name> <etal/></person-group>., vol. <volume>11864</volume> (Doha, Qatar: <publisher-name>Springer International Publishing</publisher-name>), <fpage>269</fpage>&#x2013;<lpage>276</lpage>.</citation></ref>
<ref id="ref46"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Niraula</surname><given-names>N. B.</given-names></name> <name><surname>Dulal</surname><given-names>S.</given-names></name> <name><surname>Koirala</surname><given-names>D.</given-names></name></person-group> (<year>2021</year>). &#x201C;Offensive Language Detection in Nepali Social Media.&#x201D; in <italic>Proceedings of the 5th Workshop on Online Abuse and Harms (WOAH 2021)</italic>. pp. 67&#x2013;75.</citation></ref>
<ref id="ref47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Omar</surname><given-names>A.</given-names></name> <name><surname>Mahmoud</surname><given-names>T. M.</given-names></name> <name><surname>Abd-El-Hafeez</surname><given-names>T.</given-names></name> <name><surname>Mahfouz</surname><given-names>A.</given-names></name></person-group> (<year>2021</year>). <article-title>Multi-label Arabic text classification in online social networks</article-title>. <source>Inf. Syst.</source> <volume>100</volume>:<fpage>101785</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.is.2021.101785</pub-id></citation></ref>
<ref id="ref48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rachidi</surname><given-names>R.</given-names></name> <name><surname>Ouassil</surname><given-names>M. A.</given-names></name> <name><surname>Errami</surname><given-names>M.</given-names></name> <name><surname>Cherradi</surname><given-names>B.</given-names></name> <name><surname>Hamida</surname><given-names>S.</given-names></name> <name><surname>Silkan</surname><given-names>H.</given-names></name></person-group> (<year>2023</year>). <article-title>Classifying toxicity in the Arabic Moroccan dialect on Instagram: a machine and deep learning approach</article-title>. <source>Indones. J. Electr. Eng. Comput. Sci.</source> <volume>31</volume>:<fpage>588</fpage>. doi: <pub-id pub-id-type="doi">10.11591/ijeecs.v31.i1.pp588-598</pub-id></citation></ref>
<ref id="ref49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rosenbaum</surname><given-names>G. M.</given-names></name></person-group> (<year>2019</year>). <article-title>Curses, insults and taboo words in Egyptian Arabic: in daily speech and in written literature</article-title>. <source>Romano-Arabica</source> <volume>19</volume>, <fpage>153</fpage>&#x2013;<lpage>188</lpage>.</citation></ref>
<ref id="ref50"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Sap</surname><given-names>M.</given-names></name> <name><surname>Card</surname><given-names>D.</given-names></name> <name><surname>Gabriel</surname><given-names>S.</given-names></name> <name><surname>Choi</surname><given-names>Y.</given-names></name> <name><surname>Smith</surname><given-names>A. N.</given-names></name></person-group> (<year>2019</year>). The risk of racial bias in hate speech detection. ACL.</citation></ref>
<ref id="ref51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schick</surname><given-names>T.</given-names></name> <name><surname>Sch&#x00FC;tze</surname><given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>Exploiting cloze questions for few shot text classification and natural language inference</article-title>. <source>arXiv preprint arXiv</source>:<fpage>2001.07676</fpage>.</citation></ref>
<ref id="ref52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shannag</surname><given-names>F.</given-names></name> <name><surname>Hammo</surname><given-names>B. H.</given-names></name> <name><surname>Faris</surname><given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>The design, construction and evaluation of annotated Arabic cyberbullying corpus</article-title>. <source>Educ. Inf. Technol.</source> <volume>27</volume>, <fpage>10977</fpage>&#x2013;<lpage>11023</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10639-022-11056-x</pub-id>, PMID: <pub-id pub-id-type="pmid">35502160</pub-id></citation></ref>
<ref id="ref53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shannaq</surname><given-names>F.</given-names></name> <name><surname>Hammo</surname><given-names>B.</given-names></name> <name><surname>Faris</surname><given-names>H.</given-names></name> <name><surname>Castillo-Valdivieso</surname><given-names>P. A.</given-names></name></person-group> (<year>2022</year>). <article-title>Offensive language detection in Arabic social networks using evolutionary-based classifiers learned from fine-tuned embeddings</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>75018</fpage>&#x2013;<lpage>75039</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3190960</pub-id></citation></ref>
<ref id="ref54"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tunstall</surname><given-names>L.</given-names></name> <name><surname>Reimers</surname><given-names>N.</given-names></name> <name><surname>Jo</surname><given-names>U. E. S.</given-names></name> <name><surname>Bates</surname><given-names>L.</given-names></name> <name><surname>Korat</surname><given-names>D.</given-names></name> <name><surname>Wasserblat</surname><given-names>M.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Efficient few-shot learning without prompts</article-title>. <source>arXiv preprint arXiv</source>:<fpage>2209.11055</fpage>.</citation></ref>
<ref id="ref55"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Urrutia Zubikarai</surname><given-names>A.</given-names></name></person-group> (<year>2020</year>). Appled NLP and ML for the detection of inappropiarte text in a communications platform, Universitat Polit&#x00E8;cnica de Catalunya.</citation></ref>
<ref id="ref56"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>Y.</given-names></name> <name><surname>Kordi</surname><given-names>Y.</given-names></name> <name><surname>Mishra</surname><given-names>S.</given-names></name> <name><surname>Liu</surname><given-names>A.</given-names></name> <name><surname>Smith</surname><given-names>N. A.</given-names></name> <name><surname>Khashabi</surname><given-names>D.</given-names></name> <etal/></person-group>. (<year>2022</year>). Self-instruct: aligning language models with self-generated instructions. arXiv preprint arXiv:2212.10560.</citation></ref>
</ref-list>
</back>
</article>