<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">48003</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2024.048003</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>A Machine Learning Approach to Cyberbullying Detection in Arabic Tweets</article-title>
<alt-title alt-title-type="left-running-head">A Machine Learning Approach to Cyberbullying Detection in Arabic Tweets</alt-title>
<alt-title alt-title-type="right-running-head">A Machine Learning Approach to Cyberbullying Detection in Arabic Tweets</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Musleh</surname><given-names>Dhiaa</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Rahman</surname><given-names>Atta</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><email>aaurrahman@iau.edu.sa</email></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Alkherallah</surname><given-names>Mohammed Abbas</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Al-Bohassan</surname><given-names>Menhal Kamel</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Alawami</surname><given-names>Mustafa Mohammed</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Alsebaa</surname><given-names>Hayder Ali</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Alnemer</surname><given-names>Jawad Ali</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>Al-Mutairi</surname><given-names>Ghazi Fayez</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-9" contrib-type="author">
<name name-style="western"><surname>Aldossary</surname><given-names>May Issa</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-10" contrib-type="author">
<name name-style="western"><surname>Aldowaihi</surname><given-names>Dalal A.</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-11" contrib-type="author">
<name name-style="western"><surname>Alhaidari</surname><given-names>Fahd</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Department of Computer Science, College of Computer Science and Information Technology, Imam Abdulrahman Bin Faisal University</institution>, <addr-line>P.O. Box 1982, Dammam, 31441</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-2"><label>2</label><institution>Department of Computer Information Systems, College of Computer Science and Information Technology, Imam Abdulrahman Bin Faisal University</institution>, <addr-line>P.O. Box 1982, Dammam, 31441</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-3"><label>3</label><institution>Department of Networks and Communications, College of Computer Science and Information Technology, Imam Abdulrahman Bin Faisal University</institution>, <addr-line>P.O. Box 1982, Dammam, 31441</addr-line>, <country>Saudi Arabia</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Atta Rahman. Email: <email>aaurrahman@iau.edu.sa</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2024</year></pub-date>
<pub-date date-type="pub" publication-format="electronic"><day>18</day><month>7</month><year>2024</year></pub-date>
<volume>80</volume>
<issue>1</issue>
<fpage>1033</fpage>
<lpage>1054</lpage>
<history>
<date date-type="received">
<day>24</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>28</day>
<month>5</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2024 Musleh et al.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Musleh et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_48003.pdf"></self-uri>
<abstract>
<p>With the rapid growth of internet usage, a new situation has been created that enables practicing bullying. Cyberbullying has increased over the past decade, and it has the same adverse effects as face-to-face bullying, like anger, sadness, anxiety, and fear. With the anonymity people get on the internet, they tend to be more aggressive and express their emotions freely without considering the effects, which can be a reason for the increase in cyberbullying and it is the main motive behind the current study. This study presents a thorough background of cyberbullying and the techniques used to collect, preprocess, and analyze the datasets. Moreover, a comprehensive review of the literature has been conducted to figure out research gaps and effective techniques and practices in cyberbullying detection in various languages, and it was deduced that there is significant room for improvement in the Arabic language. As a result, the current study focuses on the investigation of shortlisted machine learning algorithms in natural language processing (NLP) for the classification of Arabic datasets duly collected from Twitter (also known as X). In this regard, support vector machine (SVM), Na&#x00EF;ve Bayes (NB), Random Forest (RF), Logistic regression (LR), Bootstrap aggregating (Bagging), Gradient Boosting (GBoost), Light Gradient Boosting Machine (LightGBM), Adaptive Boosting (AdaBoost), and eXtreme Gradient Boosting (XGBoost) were shortlisted and investigated due to their effectiveness in the similar problems. Finally, the scheme was evaluated by well-known performance measures like accuracy, precision, Recall, and F1-score. Consequently, XGBoost exhibited the best performance with 89.95% accuracy, which is promising compared to the state-of-the-art.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Supervised machine learning</kwd>
<kwd>ensemble learning</kwd>
<kwd>cyberbullying</kwd>
<kwd>Arabic tweets</kwd>
<kwd>NLP</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Cyberbullying is defined as a person or group using telecommunication and digital devices to threaten others via communication networks. It includes verbal abuse, harassment, and aggressive words. Cyberbullying may denigrate and unfairly criticize someone or impersonate others&#x2019; identities. It has harmed many individuals worldwide, particularly teenagers, as it gets more accessible on widespread social networking sites [<xref ref-type="bibr" rid="ref-1">1</xref>]. Cyberbullying is the act of bullying on the internet, so there is no physical damage, but the victims express psychological harm. People will be rude behind the screen, and they are anonymous, so there is no liability for the bully. The number of people bullying the victim is enough to affect him/her with the words. Because in real life, one is driven by a limited number, but on the internet, it is up to thousands [<xref ref-type="bibr" rid="ref-2">2</xref>]. &#x201C;One in five parents around the world say their child has experienced cyberbullying at least once in 2018&#x201D; [<xref ref-type="bibr" rid="ref-3">3</xref>]. Cyberbullying is rising every day so that the numbers might become more extensive. It is, therefore, essential to prevent this by first detection, and then remedial actions may be taken employing cyber laws. Nowadays, most use Twitter (also known as X) to communicate and express ideas. Twitter has its own rules to stop the horrible act, whether it is cyberbullying or other things. However, the problem is that it needs to monitor 192 million [<xref ref-type="bibr" rid="ref-4">4</xref>] when you want to monitor up to millions of users and use the site 24 h a day and seven days a week. So, humans cannot do this monitoring manually. On the other hand, Machine Learning (ML) has become practical and can do the job without human intervention in real-time. Based on a survey conducted [<xref ref-type="bibr" rid="ref-4">4</xref>], several implications of cyberbullying have been noticed, as given in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>. The first Category is about those who experienced cyberbullying. The second Category is about its impact, and the third Category is about the type of cyberbullying that occurred to the individuals.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Cyberbullying statistics about its experience, impact, and type</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48003-fig-1.tif"/>
</fig>
<p>The survey reveals the potential frequency, harm, and kind of impact social media users have, which is worrisome. There is clear evidence and motivation to conduct studies where cyberbullying should be adequately monitored, and appropriate measures should be taken by the authorities to prevent the consequences in society. It is one of the significant areas to target in Saudi Vision 2030, which aims to transform the individual&#x2019;s lifestyle and improve well-being. In the literature, this task has been frequently done for the English language; however, for the Arabic language, studies are limited, and more research is needed, especially considering the diverse dialects in the Arabic language. Arabic is one of the most popular languages, ranking sixth globally. There are 5.2% of Arabic users on the internet [<xref ref-type="bibr" rid="ref-5">5</xref>]. Many teenagers use social media applications like Twitter, Snapchat, Instagram, and TikTok. Because of bullies, teenagers may get depressed or fear facing the world. Bullying in the past was famous in schools and is now known as cyberbullying on the internet. Many teenagers stayed at home because of the Coronavirus disease 2019 (COVID-19), spending more time online, and due to cyberbullying, faced anxiety and other psychological issues [<xref ref-type="bibr" rid="ref-5">5</xref>]. So, the situation demands a system to detect cyberbullying, which can help prevent bullying. Motivated by that, in this paper, we propose to develop a model for preprocessing Arabic tweets and detecting cyberbullying using machine learning techniques. The proposed research aims to identify cyberbullying in Arabic tweets by automatically applying machine learning algorithms with the help of Arabic Natural Language Processing (ANLP) approaches. This research is motivated by relatively few studies identifying cyberbullying that have been conducted in the Arabic language. While an overwhelming increase in the number of Arabic speaking users has been witnessed over last few years due to extended support for the native and non-English languages over the social media platforms in general and twitter in particular. This is first of its kind study in the Kingdom of Saudi Arabia. The significant contributions and aims of the study that makes it prominent from others, are listed below:
<list list-type="bullet">
<list-item>
<p>Collection and preparation of a standard dataset comprising Arabic tweets collected over the entire Arab region, especially Saudi Arabia, to comprehend diverse dialects of the language.</p></list-item>
<list-item>
<p>A comprehensive review of related literature over the past decade for cyberbullying detection in general and particularly in Arabic language.</p></list-item>
<list-item>
<p>Investigation of several machine learning approaches on the self-curated and secondary dataset from the literature.</p></list-item>
<list-item>
<p>An improvement in the results compared to the state-of-the-art techniques in the literature.</p></list-item>
</list></p>
<p>The rest of the paper is organized as follows: <xref ref-type="sec" rid="s2">Section 2</xref> provides the background of the study, and <xref ref-type="sec" rid="s3">Section 3</xref> is dedicated to related work. <xref ref-type="sec" rid="s4">Section 4</xref> contains the proposed methodology. <xref ref-type="sec" rid="s5">Section 5</xref> presents results and discussion, while <xref ref-type="sec" rid="s6">Section 6</xref> concludes the study.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Background</title>
<p>Cyberbullying is defined as any harm done constantly and intentionally through the internet. Cyberbullying has four core elements: harm, intent, repetition, and imbalance of power. Furthermore, to be considered cyberbullying, the bully must cause harm, and the damage must be caused willingly. Also, the harm must be repeated; a one-time insult will not be considered cyberbullying. In addition, the bully must have a higher power. For example, they are more prevalent [<xref ref-type="bibr" rid="ref-6">6</xref>]. Another study shows an increase in cyberbullying, as 56.1% of 351 users have admitted to being affected by cyberbullying. The study also demonstrates that the adverse effects of cyberbullying, fear, powerlessness, anger, anxiety, and sadness are the same as those of face-to-face bullying [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-8">8</xref>]. Since cyberbullying detection mainly relies on NLP algorithms, subsections focus on the commonly used steps to provide a background.</p>
<sec id="s2_1">
<label>2.1</label>
<title>Natural Language Processing (NLP)</title>
<p>Humans use Natural language to communicate, whereas computers only analyze zeros and ones [<xref ref-type="bibr" rid="ref-8">8</xref>]. Technology today has advanced, so we can make computers study languages by transferring them to their understanding using NLP, which assists computers in understanding human languages. Nowadays, we use NLP a lot to communicate with computers and devices daily, for example, language translators. However, NLP is an old concept that was proposed in [<xref ref-type="bibr" rid="ref-9">9</xref>]. Then, in 1954, Georgetown University and IBM worked together to translate Russian sentences into English. It was an automatic system capable of high-quality translation of 250 words and six &#x201C;grammar&#x201D; criteria [<xref ref-type="bibr" rid="ref-10">10</xref>].</p>
<sec id="s2_1_1">
<label>2.1.1</label>
<title>Tokenization</title>
<p>Tokenization is breaking down a phrase, sentence, paragraph, or even text document into smaller pieces like individual words or concepts. Tokens are the names given to each of these smaller units. Because Tokenization decreases word typographical variance, it is essential for the feature extraction and bag of words (BoW) procedures. Consequently, the words are transformed into features using a feature dictionary, vectors, or feature index. The feature&#x2019;s index is the frequency of (expression) in the vocabulary related to its usage [<xref ref-type="bibr" rid="ref-11">11</xref>]. Further, Tokenization can be used to count numbers and the words frequency.</p>
</sec>
<sec id="s2_1_2">
<label>2.1.2</label>
<title>Stemming</title>
<p>Stemming is the process of reducing a word to its word stem, which affixes to suffixes and prefixes or the word&#x2019;s roots, known as a lemma. The Arabic language is complicated compared to other languages in stemming techniques. <xref ref-type="table" rid="table-1">Table 1</xref> shows some examples of stemming by removing its prefixes and suffixes. It may also consider infixes.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Prefix, suffix, and infix example</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Word</th>
<th>Letter(s)</th>
<th>Stem</th>
<th>Type of affix</th>
</tr>
</thead>
<tbody>
<tr>
<td><inline-graphic xlink:href="CMC_48003-inline-1.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-8.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-15.tif"/></td>
<td>Prefix</td>
</tr>
<tr>
<td><inline-graphic xlink:href="CMC_48003-inline-2.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-9.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-16.tif"/></td>
<td>Prefix</td>
</tr>
<tr>
<td><inline-graphic xlink:href="CMC_48003-inline-3.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-10.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-17.tif"/></td>
<td>Prefix</td>
</tr>
<tr>
<td><inline-graphic xlink:href="CMC_48003-inline-4.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-11.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-18.tif"/></td>
<td>Infix</td>
</tr>
<tr>
<td><inline-graphic xlink:href="CMC_48003-inline-5.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-12.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-19.tif"/></td>
<td>Suffix</td>
</tr>
<tr>
<td><inline-graphic xlink:href="CMC_48003-inline-6.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-13.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-20.tif"/></td>
<td>Suffix</td>
</tr>
<tr>
<td><inline-graphic xlink:href="CMC_48003-inline-7.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-14.tif"/></td>
<td><inline-graphic xlink:href="CMC_48003-inline-21.tif"/></td>
<td>Suffix</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Kanan et al. [<xref ref-type="bibr" rid="ref-12">12</xref>] mentioned their experience using stemming to reduce the number of words in the text. By using the stem, all forms of a word (name, adverb, adjective, or character) are discarded by returning the word to its original root. The study focused on removing the prefixes since the authors believed it would make document classification more effective. They used three NLP preprocessing tools for detection: normalization, stop word removal and stemming. Then, they applied a set of machine learning algorithms, for classification: K-nearest neighbors (KNN), SVM, NB, RF, and J48. When compared with and without stemming, it was found that stemming increased accuracy and F1-score.</p>
</sec>
<sec id="s2_1_3">
<label>2.1.3</label>
<title>Removing Stop Words</title>
<p>Stop words are used as unnecessary filler words for the sentence; removing them is essential. For example, stop words in English are the &#x2018;A,&#x2019; &#x2018;Is,&#x2019; &#x2018;The,&#x2019; and &#x2018;Are,&#x2019; and there are many more. The study focuses on removing the Arabic stop words. These words have been prepared; some are in <xref ref-type="table" rid="table-2">Table 2</xref>. For the Arabic Language, Abu El-Khair [<xref ref-type="bibr" rid="ref-13">13</xref>] defined a list that contains 56 Arabic stop words.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Example of Arabic stop words</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>#</th>
<th>Word</th>
</tr>
</thead>
<tbody>
<tr>
<td>1</td>
<td><inline-graphic xlink:href="CMC_48003-inline-22.tif"/></td>
</tr>
<tr>
<td>2</td>
<td><inline-graphic xlink:href="CMC_48003-inline-23.tif"/></td>
</tr>
<tr>
<td>3</td>
<td><inline-graphic xlink:href="CMC_48003-inline-24.tif"/></td>
</tr>
<tr>
<td>4</td>
<td><inline-graphic xlink:href="CMC_48003-inline-25.tif"/></td>
</tr>
<tr>
<td>5</td>
<td><inline-graphic xlink:href="CMC_48003-inline-26.tif"/></td>
</tr>
<tr>
<td>6</td>
<td><inline-graphic xlink:href="CMC_48003-inline-27.tif"/></td>
</tr>
<tr>
<td>7</td>
<td><inline-graphic xlink:href="CMC_48003-inline-28.tif"/></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_1_4">
<label>2.1.4</label>
<title>N-Gram</title>
<p>N-gram is a series of letters that, when combined, will help understand their meaning. We use N-gram to spell sentences as every word, two words [<xref ref-type="bibr" rid="ref-14">14</xref>]. Following are examples of different N-grams: &#x201C;My laptop&#x201D; (2-grams), &#x201C;I have kids&#x201D; (3-grams), and &#x201C;I am a student&#x201D; (4-grams). There are three purposes for using N-gram: first, correcting spelling mistakes. For example, when &#x201C;Los Angeles&#x201D; is written as &#x201C;Los Angeles,&#x201D; N-gram can predict and fix errors. Second, it can choose which words can be grouped. For example, it can indicate that &#x201C;Los&#x201D; and &#x201C;Angeles&#x201D; can be combined as &#x201C;Los Angeles.&#x201D; Finally, it can assist in anticipating the next word to see the whole meaning if the user erases words like &#x201C;I drink&#x201D; and can predict the next word, &#x201C;water or Juice&#x201D; [<xref ref-type="bibr" rid="ref-15">15</xref>].</p>
</sec>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Related Work</title>
<p>This section will summarize cyberbullying-related studies by reviewing the preprocessing methods applied in each research, the classifiers used, the dataset provided, the features set, and the results.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Data Collection</title>
<p>This section presents the different approaches for data collection. First, a group of authors obtained their data primarily from Twitter and supplemented it with other sources. For example, Al-Ajlan et al. [<xref ref-type="bibr" rid="ref-16">16</xref>] received their dataset from Twitter utilizing the Twitter streaming API and querying with a bad word list, and the total number of tweets is 39,000. Haider et al. [<xref ref-type="bibr" rid="ref-17">17</xref>,<xref ref-type="bibr" rid="ref-18">18</xref>] gathered their dataset from Twitter and Facebook using two custom-built tools: a Twitter scraper written in PHP and a Facebook scraper written in Python. In addition, the tools were linked to the mango database server to save data. In [<xref ref-type="bibr" rid="ref-19">19</xref>], the data was gathered by AlHarbi et al. from Twitter through the Twitter API, Microsoft-FLOW, and YouTube comments and was then compiled into a single file comprising 100,327 tweets and comments. Mouheb et al. [<xref ref-type="bibr" rid="ref-20">20</xref>], a dataset containing 25,000 comments and tweets, was gathered from Twitter and YouTube using the YouTube Data API and the Twitter API, respectively. Haider et al. [<xref ref-type="bibr" rid="ref-21">21</xref>] gathered a data collection size of 34,890 from Twitter using a program created by the author team. Kanan et al. [<xref ref-type="bibr" rid="ref-12">12</xref>] gathered their dataset from Twitter using RStudio and Stool, statistical and mathematical tools for extracting tweets. In addition, the dataset contains 19,650/6138 tweets. Phanomtip et al. [<xref ref-type="bibr" rid="ref-22">22</xref>] detected cyberbullying in the Twitter dataset, and the hate speech tweet dataset was extracted from a paper study, which contained 38,686 and 68,519 tweets, respectively. In [<xref ref-type="bibr" rid="ref-23">23</xref>], the dataset was collected by Banerjee et al. from Twitter, and the size is 69,874. Likewise, Bharti et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] collected a mixed dataset of two public datasets gathered via Twitter API from Twitter. The dataset collected by Almutiry et al. [<xref ref-type="bibr" rid="ref-11">11</xref>] consisted of 17,748 tweets obtained from Twitter through the Twitter API and Arabi Tools. Lokhande et al. [<xref ref-type="bibr" rid="ref-25">25</xref>] also used Twitter as a dataset because it generates data daily. Almutairi et al. [<xref ref-type="bibr" rid="ref-26">26</xref>], after signing in to Twitter via the developer&#x2019;s app, created a new application. They generated access tokens and access token secrets, using Python programming language to access the Twitter API via an open-source library package called tweety. Jain et al. [<xref ref-type="bibr" rid="ref-27">27</xref>] utilized a combination of several datasets containing hate speech: the hate speech Twitter dataset, the hate speech language dataset with tweets, and the Wikipedia dataset. In [<xref ref-type="bibr" rid="ref-28">28</xref>], a Java application was created by Al-Mamun et al. to extract social media data. Twitter and Facebook were proposed for Bangla text. Using Facebook Graph API and Twitter Rest API, 1000 pieces were collected from Facebook, and 1400 Bangla public status updates were obtained from Twitter. Authors in [<xref ref-type="bibr" rid="ref-29">29</xref>] used Twitter API to collect data, then preprocessed it with the NL Toolkit (NLTK), and the term frequency&#x2013;inverse document frequency (TF-IDF) vectorizer performed the extraction. Balakrishnan et al. [<xref ref-type="bibr" rid="ref-30">30</xref>], using the GamerGate hashtag, collected data from Twitter and used the cyberbullying dataset provided by authors in [<xref ref-type="bibr" rid="ref-31">31</xref>] gathered the information from three distinct social networks: Twitter, Formspring, and Wikipedia, each focusing on a different aspect of cyberbullying&#x2014;racism and misogyny on Twitter, bullying on Formspring, and Wikipedia attacks.</p>
<p>The second group of authors collected data from sources other than Twitter in their fundamental research. Di-Capua et al.&#x2019;s [<xref ref-type="bibr" rid="ref-32">32</xref>] dataset was published in public, and the data was taken from Formspring, me, and YouTube. Alakrot et al. [<xref ref-type="bibr" rid="ref-33">33</xref>] also used YouTube; their method was to select some YouTube channels, upload controversial videos about Arab celebrities, and collect comments from them. The data set collected about 15,050 comments. Several authors, like [<xref ref-type="bibr" rid="ref-2">2</xref>], who collected Formspring data, namely Me and MySpace, used Formspring. Reynolds et al. [<xref ref-type="bibr" rid="ref-34">34</xref>] also collected the data from Formspring. The authors selected random files from the 18,554 users and then used Amazon&#x2019;s Mechanical Turk service to determine the labels for the truth sets. Maral et al. [<xref ref-type="bibr" rid="ref-35">35</xref>] also used three datasets in their study: Formspring (a Q&#x0026;A forum), Wikipedia talk pages (a collaborative knowledge repository), and Twitter (a microblogging platform). All these datasets are manually labeled and publicly available. In a study by Hani et al. [<xref ref-type="bibr" rid="ref-36">36</xref>], the Kaggle dataset was initially taken from a research paper from Formspring and contains 12,773 conversations. Kaggle also has been used by Srivastava et al. [<xref ref-type="bibr" rid="ref-37">37</xref>]. The last group of authors used multiple and different ways and datasets for their data collection. Rachid et al. [<xref ref-type="bibr" rid="ref-38">38</xref>] collected the dataset from <ext-link ext-link-type="uri" xlink:href="http://Aljazeera.net">Aljazeera.net</ext-link> (accessed on 10/05/2024), and the comments were selected using CrowdFlower.</p>
<p>Furthermore, the comments were classified into three categories: Obscene (533 comments), offensive (25,506 comments), and clean (5653); the data by Husain in [<xref ref-type="bibr" rid="ref-39">39</xref>] was provided by the shared task of the fourth workshop on Open-Source Arabic Corpora and Corpora Processing tools (OSACT) in Language Resource and Evaluation Conference (LREC) 2020. Furthermore, the data was released into three different data parts: training dataset (1000 tweets), development dataset (1367 tweets), and testing dataset (5468). In [<xref ref-type="bibr" rid="ref-40">40</xref>], the dataset used by Alfageh et al. was taken from a publicly available dataset that contains 15,000 comments from YouTube. In [<xref ref-type="bibr" rid="ref-41">41</xref>], the dataset had 2218 sessions provided by a research report collected from Instagram using snowball sampling. Yin et al. [<xref ref-type="bibr" rid="ref-42">42</xref>] used data from three datasets: Kongregate, Slashdot, and MySpace. Then, they manually labeled a randomly selected subset of the threads from each labeled dataset. A study in [<xref ref-type="bibr" rid="ref-43">43</xref>] provided a semi-synthetic, flexible, and scalable dataset to mitigate shortcomings of current cyberbullying detection datasets. The dataset covers various traits such as hostility, reprise, aim to hurt, and issues among peers. &#x00D6;zel et al. [<xref ref-type="bibr" rid="ref-44">44</xref>] manually collected around 900 messages from Twitter and Instagram. However, only half of them contained cyberbullying terms. Dadvar et al. [<xref ref-type="bibr" rid="ref-45">45</xref>] collected cyberbullying related data from MySpace and Fundacion Barcelona Media. Similarly, Buan et al. [<xref ref-type="bibr" rid="ref-46">46</xref>] used the third version of the bullying traces dataset that was created by Professor Zhu at the University of Wisconsin-Madison. Authors in [<xref ref-type="bibr" rid="ref-47">47</xref>] collected their data from Perverted-Justice (PJ), an American organization investigates, identifies, and publicizes the conduct of adults who solicit online sexual conversations with adults posing as minors.</p>
<p>Based on the brief review of cyberbullying detection datasets, it is apparent that most of the data has been collected, processed, and annotated manually by the researchers. Most of the data sources are comprised of English language while for Arabic language datasets are limited. Moreover, Twitter and Instagram are the most famous targeted social media.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Review of Machine Learning Algorithms in NLP</title>
<sec id="s3_2_1">
<label>3.2.1</label>
<title>Support Vector Machine (SVM)</title>
<p>SVM is a fast and dependable classification algorithm. Supervised machine learning has been shown to give excellent and accurate performance results. This section summarizes the primary use of the SVM algorithm for word classification, focusing on cyberbullying detection.</p>
<p>Phanomtip et al. [<xref ref-type="bibr" rid="ref-22">22</xref>] start by filtering the tweets by removing every unimportant information like tags (@), URLs, and locations. Then, they used embedding methods: TF-IDF and document to vector (Doc2Vec) to extract the vector from the filtering tweet and feed it into the linear SVM for the classification. With a large dataset of 68 K tweets, the results state that the experiments for the linear SVM gave excellent accuracy in detecting cyberbullying, with 91% and 86% accuracy using TF-IDF and Doc2Vec, respectively. For a small dataset, Hani et al. [<xref ref-type="bibr" rid="ref-36">36</xref>] used 1.6 K posts and applied the TF-IDF method to extract the vector for the linear SVM. The result was also reasonable accuracy, more than 89%. Buan et al. in [<xref ref-type="bibr" rid="ref-46">46</xref>] followed the same approach by applying the linear SVM to distinguish between bullying and other texts. The idea was to weigh each term by giving it a gram scale. The accuracy of a dataset consisting of more than 14 K tweets was about 77%. Likewise, studies in [<xref ref-type="bibr" rid="ref-47">47</xref>,<xref ref-type="bibr" rid="ref-48">48</xref>] used the same feature and the SVM to reduce the number of parts by weighing their importance to the class attribute.</p>
</sec>
<sec id="s3_2_2">
<label>3.2.2</label>
<title>Na&#x00EF;ve Bayes (NB)</title>
<p>NB technique is a simple text categorization algorithm. It is a probabilistic approach for each attribute in each class set. It has been effectively used for various issues and applications but excels in NLP. The studies using NB as their main algorithm are from Rachid et al. in [<xref ref-type="bibr" rid="ref-38">38</xref>], Mouheb et al. in [<xref ref-type="bibr" rid="ref-20">20</xref>], Kanan et al. in [<xref ref-type="bibr" rid="ref-12">12</xref>], Haider et al. in [<xref ref-type="bibr" rid="ref-18">18</xref>], Al-Mamun et al. in [<xref ref-type="bibr" rid="ref-28">28</xref>] and Agrawal et al. in [<xref ref-type="bibr" rid="ref-31">31</xref>]. They used the NB model and collected their data set from Twitter but slightly differed in the preprocessing step and methodology. First, Rachid et al. [<xref ref-type="bibr" rid="ref-38">38</xref>] eliminated Arabic and English punctuation, HTML codecs, numbers and symbols, and words of size one. In addition, diacritics and normalization of Arabic text are all part of the preprocessing stages. In addition, the models used are NBM machine learning and Bag of a Word, which resulted in 87% Presisions, 35% Recall, and 50% F1-score. Second, Mouheb et al. [<xref ref-type="bibr" rid="ref-20">20</xref>], the preprocessing steps include removing Arabic diacritics, Arabic determiner, and normalizations. Furthermore, using the NB model resulted in 95% accuracy. Third, Kanan et al. in [<xref ref-type="bibr" rid="ref-12">12</xref>], the preprocessing steps removed diacritics, non-Arabic literals, symbols, normalization, stop words, stemming, and 91% accuracy. Fourth, Al-Mamun et al. [<xref ref-type="bibr" rid="ref-28">28</xref>] investigated the NB approach for cyberbullying detection in users&#x2019; posts in Bangla text. The NB scored an F1-score equal to 69% and achieved an accuracy of 60.98%. In English language text, however, the F1-score was 39%, and the accuracy was 40.98%. Finally, Agrawal et al. in [<xref ref-type="bibr" rid="ref-31">31</xref>] investigated the NB approach with character N-grams as a feature selection method. Consequently, the F1-score for the NB was 35.9% for bullying detection, 68.6% for racism on the Formspring dataset, 64.7% for sexism on the Twitter dataset, and 66.5% for attack-related text in Wikipedia. For word unigrams, NB F1-scores were 0.025 for bullying in Formspring, 0.617 for racism, 0.635 for sexism on Twitter, and 0.659 for the attack on Wikipedia. Alfageh et al. in [<xref ref-type="bibr" rid="ref-40">40</xref>] and &#x00D6;zel et al. in [<xref ref-type="bibr" rid="ref-44">44</xref>] utilized the NBM. To begin with, the data set was obtained from YouTube, and the data is balanced; the preprocessing process includes the removal of all non-Arabic literals, symbols punctuation, URLs, hashtags and Arabic diacritics, normalization, and stemming. The model employed is NBM, which has an accuracy of 78.4%. Consequently, &#x00D6;zel et al. [<xref ref-type="bibr" rid="ref-44">44</xref>], in their research on detecting cyberbullying for the Turkish language NBM, were the most successful in their experiments for running time and accuracy. SVM followed them in second place. On the other hand, even though instance-based K-nearest neighbors (IBK) have no training, their running duration is lengthy. J48, on the other hand, has a relatively short testing time but a pervasive training time. As a result, NBM was the top performer. On the other hand, the studies from Bharti et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] performed the worst among all the classifiers and NB algorithms. The accuracy was 66.86%, with an F1-score of 68.18% in cyberbullying detection. Further, the decree showed an accuracy of 90.59% and an F1-score of 92.79%.</p>
</sec>
<sec id="s3_2_3">
<label>3.2.3</label>
<title>SVM and NB</title>
<p>In their study, authors [<xref ref-type="bibr" rid="ref-12">12</xref>] discussed supervised machine learning techniques in Arabic social media content for detecting cyberbullying. They rely on the classifying process using several classifiers: SVM and NB. They start by stating their dataset of 14 K tweets collected from Twitter using RStudio and Rtool and 2 K posts from Facebook. Afterward, they used the Waikato Environment for Knowledge Analysis (WEKA) Toolkit to implement preprocessing tools like Stop-Word-Removal and Stemming. They conclude that SVM gives a better result when the size of the datasets increases, with an F-measure of more than 90%. At the same time, NB gave 10% fewer results than SVM, with an F-measure of less than 85%. Haider et al. [<xref ref-type="bibr" rid="ref-18">18</xref>] used the same approach with a massive dataset of 126K posts collected from the same sources (Twitter and Facebook). The results also favored SVM with an F-measure of more than 92%, while NB achieved an F-measure of 90%. Rachid et al. [<xref ref-type="bibr" rid="ref-38">38</xref>] provided a dataset of 32 K collected from <ext-link ext-link-type="uri" xlink:href="https://Aljazeera.net">Aljazeera.net</ext-link>, considering if it is a balanced or unbalanced dataset. They applied two feature extraction methods: TD-IDF and BoW. The result shows that SVM is better than NB regardless of whether the dataset is balanced or unbalanced. SVM gave a higher detection accuracy when applying the TD-IDF method, while NB used the Bow method. Haider et al. [<xref ref-type="bibr" rid="ref-21">21</xref>] provided almost the exact dataset size as Rachid et al., but they collected it from Twitter. Also, they applied two different feature extraction methods, Boosting and Bagging. They conclude that SVM beat NB by about 0.4% accuracy higher for both ways. Atoum [<xref ref-type="bibr" rid="ref-48">48</xref>] proposed a simulated annealing (SA) model for the detection process. He used SVM and NB as supervised machine-learning classification tools for this model. He provides a dataset of 5K tweets collected from Twitter API. The experiment results indicated that SVM classifiers have outperformed NB classifiers as they achieved an average accuracy value of more than 90%, while NB gained about 81%. Bharti et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] followed the same approach but with a large dataset of 57 K tweets. The results were similar: SVM accuracy was 91.48%, and NB was 91.35%. From the review, we can conclude that SVM is a promising supervised ML model for classification regardless of the dataset size. While NB significantly lacks accuracy when the dataset decreases [<xref ref-type="bibr" rid="ref-49">49</xref>]. Such as, Dalvi et al. [<xref ref-type="bibr" rid="ref-29">29</xref>] used a minimal dataset; the results were about 71% for the SVM model and 52% for the NB model.</p>
</sec>
<sec id="s3_2_4">
<label>3.2.4</label>
<title>Random Forest</title>
<p>A random forest (RF) comprises many discrete decision trees that perform with a forest. Each tree in the RF generates a class prediction, and the class with the most votes becomes the model&#x2019;s prediction. Their performance is strict to match other algorithms because the trees shield each other from their faults. While some trees may be incorrect, many others will be correct. They can also handle various feature types, including binary, categorical, and numerical. During review RF was used in 7 out of 36 research. There is only one paper used alone. The others used it with other classifiers to compare RF with the others. Al-Ajlan et al. [<xref ref-type="bibr" rid="ref-16">16</xref>] used RF to detect cyberbullying because of its prominence in this field. The model was based on the user&#x2019;s personality and is determined by the Big Five and Dark Triad models. Psychopathy 91.8% had the best RF performance in Dark Triad while the others followed by a small margin of 2%. The Big Five had RF higher than the rest, with almost 91% Neuroticism, Extraversion, and Agreeableness, followed by 90%. So, when the lower RF was 90%, their research shows that the Big Five and Dark Triad models were proven to enhance RF in cyberbullying detection considerably. In three of seven of the research, Husain [<xref ref-type="bibr" rid="ref-39">39</xref>], Kanan, et al. [<xref ref-type="bibr" rid="ref-12">12</xref>], and Agrawal et al. [<xref ref-type="bibr" rid="ref-31">31</xref>], RF was the second-highest performing model. The first research is Husain [<xref ref-type="bibr" rid="ref-39">39</xref>], which detects offensive Arabic language from a dataset of tweets using machine learning models in groups (AdaBoost, Bagging, and RF). Count and TF-IDF were utilized for feature extraction. Their result showed that bagging at 88% had the best F1-score performance, while the RF was the second F1-score by 87%. The second research, Kanan et al. [<xref ref-type="bibr" rid="ref-12">12</xref>], aims to identify such harmful written posts; they proposed using Machine Learning. They used a variety of classifier algorithms (KNN, SVM, NB, RF, and Decision Trees (also known as J48). The first method involved testing each classifier with all the preprocessing stages; RF had a higher F-measure of 94.49%. The second method involved testing the classifiers without stemming; RF had the first F1-score of 94.4%. The third method was to try the classifiers without removing stop words. RF had the best F-measure of 93.4%. The fourth method was to compare the performance of the classifiers with all preprocessing stages on Facebook data only, followed by Twitter records, where SVM had the best F-measure of 94.4%, then RF followed by 91.4% F-measure on Twitter. While on Facebook, SVM had the first F-measure by 91.7l%, while RF was the second by 94.1% F-measure. Remaining studies had RF as the third-best F-measure Rachid et al. [<xref ref-type="bibr" rid="ref-38">38</xref>], Bharti et al. [<xref ref-type="bibr" rid="ref-24">24</xref>], and Jain et al. [<xref ref-type="bibr" rid="ref-27">27</xref>]. Whereas Jain et al. [<xref ref-type="bibr" rid="ref-27">27</xref>] used SVM, Logistic Regression (LR), RF, and Multi-Layered Perceptron (MLP) for classification. Three feature extraction methods were employed: the first technique was the BoW model, second with TF-IDF and last is word to vector (word2vec). The greatest F-measure value for the Twitter dataset is obtained as 93.4% when utilizing BoW and LR. The second F-measure was RF, with 93.4%. The TF-IDF&#x2019;s best F-measure was SVC at 93.9%, followed by LR at 93.6%, and then the third F-measure was RF at 93.3%. The best F1-score in word2vec was MLP at 92.2%; RF was the second F-measure at 92.2%. The best F-measure for the Wikipedia dataset was 83.7 % when TF-IDF and SVC were employed after LR at 82.5%, then RF at 81.8%. Then, BoW was the second-best F-measure. Their best was LR at 82.9%; then RF was the second with 81.8% F-measure. Word2Vec had the lowest F-measures. Their highest was MLP, at 82.5%. At the same time, the lowest F-measure was RF at 77.6%. Then, Bharti et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] extracted features using BoW. They then utilized several classification methods: RF, decision tree, NB, XGBoost, SVM, and logistic regression. The best F-measure was LR, with about 94.06%. Then, SVM by 91.99%. The third is RF 91.8%. The last research was Rachid et al. [<xref ref-type="bibr" rid="ref-38">38</xref>], who gathered their dataset from the Arabic news channel Aljazeera website. The dataset is 32 K comments. Their best machine learning models had 85% accuracy; they got that with three different classifiers, RF, XGBoost and SVM with TF-IDF and N-gram.</p>
</sec>
<sec id="s3_2_5">
<label>3.2.5</label>
<title>SVM vs. RF</title>
<p>Some work has been done using SVM and RF techniques. Rachid et al. [<xref ref-type="bibr" rid="ref-38">38</xref>] used massive and imbalance Arabic datasets collected from the <ext-link ext-link-type="uri" xlink:href="https://Aljazeera.net">Aljazeera.net</ext-link> website. After they implemented the SVM algorithm, the F-measure was 98%, and with the RF algorithm, it was 80%. Bharti et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] also used the same dataset feature, but the dataset was collected from Twitter in English. The result was 93.3% for SVM and 93% for RF. Jain et al. [<xref ref-type="bibr" rid="ref-27">27</xref>] have a dataset like that of Bharti et al. [<xref ref-type="bibr" rid="ref-24">24</xref>], and they got 93.9% for SVM and 93.3% for RF. All three researchers used TF-IDF for feature extraction and found that SVM is better than RF. On the other hand, Kanan et al. [<xref ref-type="bibr" rid="ref-12">12</xref>] and Husain [<xref ref-type="bibr" rid="ref-39">39</xref>] used large and imbalance datasets in the Arabic language. They conclude that RF is better than SVM. SVM is better when the dataset is large enough, more than 20 K, and RF will be better when the dataset is less than 20 K. So, it shows that there will be no difference if the dataset is an Asian foreign language [<xref ref-type="bibr" rid="ref-50">50</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>].</p>
</sec>
<sec id="s3_2_6">
<label>3.2.6</label>
<title>Logic Regression (LR)</title>
<p>LR is a popular machine learning algorithm that uses the logistic function to classify data. Moreover, it is best used if the classification only holds two categories: bullying and non-bullying. However, it can be used in multiclass classification if one runs the algorithm multiple times using the one-versus-all technique. In most related work, logistic regression showed a higher result than other algorithms. Machine learning (ML) has been used as a deep learning (DL) baseline. Bharti et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] used LR as one of the ML models, and LR was the best overall, with an accuracy of 92%. Rachid et al. [<xref ref-type="bibr" rid="ref-38">38</xref>] have also used LR as a baseline firstly with a BoW on a dataset with more CB (cyberbullying) than NCB (non-cyber bullying) and had an F1-score of 30%, which is a very low due to the imbalance. The second time, it was used on another imbalance dataset with more NCB than CB to make the ratio realistic; the F1-score of the second time was 53%, which is a clear improvement. The third time, LR was used with character level TF-IDF N-gram on a balanced dataset, and it got an F1-score of 84%. Agrawal et al. [<xref ref-type="bibr" rid="ref-31">31</xref>] used LR with both character N-grams and word unigrams, and the highest F1-score was 72% and 76%, respectively. In addition to being used as a baseline for DL, LR has been used for ensemble machine learning. Husain in [<xref ref-type="bibr" rid="ref-39">39</xref>] got an F1-score of 81% for LR, exceeding the decision tree and being 1% lower than SVM. Haidar et al. [<xref ref-type="bibr" rid="ref-21">21</xref>] had a different purpose for LR as they used it as one of five single learners incorporated for stacking ensemble machine learning. Jain et al. [<xref ref-type="bibr" rid="ref-27">27</xref>] have experimented with three feature selection methods: BoW, TF-IDF, and word2vec on two data sets from Twitter and Wikipedia. They got the highest accuracy of 92% when LR was combined with BoW. Alfageh et al. [<xref ref-type="bibr" rid="ref-40">40</xref>] have also used TF-IDF with LR, but the F1-score was lower by 1.8% than when it was used with count vectorization, which had an F1-score of 78.6%.</p>
</sec>
<sec id="s3_2_7">
<label>3.2.7</label>
<title>Other Classifiers</title>
<p>Authors in [<xref ref-type="bibr" rid="ref-16">16</xref>] proposed a Convolution Neural Network (CNN) for cyberbullying detection by adopting profound learning principles instead of machine learning. They implemented their system by applying four steps, starting with embedding the texts as a numerical representation. Then, convolving the input vectors to detect the features and compress the output of the convolution process to smaller matrices so that only significant and transparent parts are considered. The last step includes the dense layer, which will feed all the outputs of the previous layers to all NN&#x2019;s neurons. They provided a dataset of 39K tweets collected using Twitter streaming API for the experiment. The results of an automated cyberbullying detector with no human involvement were excellent as the accuracy was more than 95% of the detection process. Banerjee et al. [<xref ref-type="bibr" rid="ref-23">23</xref>] applied the same method with a larger dataset consisting of 69K tweets, and the results were also outstanding with more than 93% accuracy. Srivastava et al. [<xref ref-type="bibr" rid="ref-37">37</xref>] performed different models of deep learning algorithms in detecting insults in social commentary, and the most important of them are Gated Recurrent Units (GRU), Long short-term memory (LSTM), and Bidirectional LSTM (BLSTM). They first applied data preprocessing to percolate the comments, including text cleaning, Tokenization, stemming, lemmatization, and stop-word removal. After that, the filtered data is passed to the deep learning algorithms for prediction. The results were very similar, as BLSTM outperformed the others by 82.18% accuracy, while GRU and LSTM achieved 81.46% and 80.86%, respectively. In comparing CNN with GRU, LSTM, and BLSTM, Benaissa et al. [<xref ref-type="bibr" rid="ref-38">38</xref>] provided a dataset of 32 K Arabic comments collected from <ext-link ext-link-type="uri" xlink:href="http://Aljazeera.net">Aljazeera.net</ext-link> for testing. From the results, it seems CNN surpasses the other models by 1% F1-score. Also, combined, they achieved an average of 84% F1-score in a balanced dataset. The KNN is a supervised ML algorithm that may be used for both classification and regression tasks [<xref ref-type="bibr" rid="ref-50">50</xref>]. However, it is mainly employed for categorization and prediction. It is defined by two characteristics: a non-parametric learning algorithm and a lazy learning algorithm. Haidar et al. [<xref ref-type="bibr" rid="ref-21">21</xref>] when they compare KNN by F1-measure to other classifiers. They had that KNN is the lowest in bagging, 90.4%, and in Boosting, 89.8%. Also, Kanan et al. [<xref ref-type="bibr" rid="ref-12">12</xref>] found that KNN has the lowest F1-score, 78.2%, and 61.9%, in Facebook and Twitter datasets. Al-Mamun et al. [<xref ref-type="bibr" rid="ref-28">28</xref>] choose those classifiers (i.e., NB, SVM, Decision Tree, and KNN). In their results, KNN was observed better than NB.</p>
<p>Based on the related work, it is evident that cyberbullying detection in Arabic language is among the hottest areas of research that need significant attention [<xref ref-type="bibr" rid="ref-52">52</xref>,<xref ref-type="bibr" rid="ref-53">53</xref>]. More studies are needed to investigate the overwhelming amount of data being produced by the social media platforms by abundant of users especially in Arabic language. Investigation of smart techniques could potentially help identify this adverse effect of social media precisely to prevent undesirable incidents. In this regard, nine well-known machine learning algorithms have been shortlisted based on their effectiveness in the similar problems observed in the literature. <xref ref-type="table" rid="table-3">Table 3</xref> presents a summary of the reviewed literature.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Summary of literature review</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Ref.</th>
<th>Classifier</th>
<th>Language</th>
<th>Dataset source and size</th>
<th>Feature extraction</th>
<th>Metric</th>
<th>Strength</th>
<th>Weakness</th>
</tr>
</thead>
<tbody>
<tr>
<td>[<xref ref-type="bibr" rid="ref-11">11</xref>]</td>
<td>SVM</td>
<td>Arabic</td>
<td>Twitter API</td>
<td>Light Stemmer,</td>
<td>Accuracy</td>
<td>N/A</td>
<td>Small</td>
</tr>
<tr>
<td/>
<td/>
<td/>
<td>17,748 tweets</td>
<td>Arabic Stemmer Khoja, TF-IDF</td>
<td> 85.49%</td>
<td/>
<td> dataset</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-12">12</xref>]</td>
<td>KNN, SVM, NB, RF, J48</td>
<td>Arabic</td>
<td>Twitter API, 4000 tweets FB 2138 posts</td>
<td></td>
<td>N/A</td>
<td>Multiple experiments</td>
<td>N/A</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-19">19</xref>]</td>
<td>Lexicon</td>
<td>Arabic</td>
<td>Twitter API,</td>
<td>The content of the</td>
<td></td>
<td>Original</td>
<td>Imbalance</td>
</tr>
<tr>
<td/>
<td>PMI</td>
<td/>
<td>YouTube,</td> 
<td>tweet</td>
<td></td>
<td>dataset,</td>
<td>data with</td>
</tr>
<tr>
<td/>
<td>Entropy</td>
<td></td>
<td>Microsoft Flow.</td>
<td></td>
<td/>
<td>lexicon</td>
<td>stop words</td>
</tr>
<tr>
<td/>
<td>Chi-square</td>
<td/>
<td>100K&#x002B;</td>
<td/>
<td/>
<td> approach.</td>
<td/>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
<td>SVM</td>
<td>English</td>
<td>Twitter API</td>
<td>TF-IDF, and</td>
<td>74%</td>
<td>Large dataset</td>
<td>Presumed</td>
</tr>
<tr>
<td/>
<td/>
<td/>
<td>67K tweets</td>
<td>Doc2Vec</td>
<td/>
<td/>
<td>sentence correction</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-24">24</xref>]</td>
<td>NB, RF, J48 XGBoost, SVM, LR</td>
<td>English</td>
<td>From old papers 57,787 tweets</td>
<td>BoW</td>
<td>Accuracy 92.60 Precision 98.73 F1-score 94.20 Recall 92.07</td>
<td>6-way classification Big dataset</td>
<td>Imbalanced dataset</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-25">25</xref>]</td>
<td>SVM, CNN Keras library</td>
<td>English</td>
<td>Twitter API</td>
<td></td>
<td>N/A</td>
<td>N/A</td>
<td>N/A</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-26">26</xref>]</td>
<td>SVM</td>
<td>Arabic</td>
<td>Twitter API 8154 tweets</td>
<td>TF-IDF</td>
<td>Accuracy 82%, Precision 81%, F1-score 82%</td>
<td>Resampling was used to balancing</td>
<td>Minimal preprocessing</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-27">27</xref>]</td>
<td>SVM, LR, RF, MLP</td>
<td>English</td>
<td>Twitter &#x0026; Wikipedia dataset 35,787 tweets 40K comments</td>
<td>BoW TF-IDF word2vec</td>
<td>Accuracy 92.1% Precision 95.9% Recall 92.7% f-measure 93.9%</td>
<td>Large dataset 4-way classifier</td>
<td>Imbalanced dataset</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-33">33</xref>]</td>
<td>NB, CNB, LR</td>
<td>Arabic</td>
<td>15,000 YouTube comments</td>
<td>TF-IDF Vectorizations</td>
<td>F1-score 78.6%</td>
<td>Large dataset and features</td>
<td>Minimal preprocessing</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-35">35</xref>]</td>
<td>CNN LSTM BLSTM SSWE</td>
<td>English</td>
<td>16K tweets 10K comments on Wikipedia 12K Q&#x0026;A</td>
<td>N/A</td>
<td>N/A</td>
<td>Many classifiers</td>
<td>Data augmentation was missing</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-37">37</xref>]</td>
<td>DL, GRU, LSTM, BLSTM, RNN</td>
<td>English</td>
<td>Kaggle dataset</td>
<td>N/A</td>
<td>Accuracy 82.12%</td>
<td>4-way classification</td>
<td>Imbalance dataset Accuracy is low</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-38">38</xref>]</td>
<td>Deep Learning</td>
<td>Arabic</td>
<td><ext-link ext-link-type="uri" xlink:href="https://Aljazeera.net">Aljazeera.net</ext-link> 32K comments</td>
<td>BoW N-grams, OOV embeddings</td>
<td>F-score 84%</td>
<td>Hybrid methods.</td>
<td>N/A</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-39">39</xref>]</td>
<td>SVM, LR, J48, RF, Bagging, AdaBoost</td>
<td>Arabic</td>
<td>Twitter API 7835 tweets</td>
<td></td>
<td>Count features, TF-IDF</td>
<td>Emoticons and emojis turned into text</td>
<td>The results representation is not clear</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-46">46</xref>]</td>
<td>SVM, LSTM, GRU</td>
<td>English</td>
<td>Bullying traces data set. 7321 tweets</td>
<td>N/A</td>
<td>Precision: 82.56% Recall: 86.42% Accuracy: 86.90%</td>
<td>N/A</td>
<td>Old and small dataset</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-48">48</xref>]</td>
<td>SVM, NB Chi-square</td>
<td>English</td>
<td>Twitter API 5628 tweets</td>
<td>N/A</td>
<td>SVM, 4-gram 92.02%</td>
<td>N/A</td>
<td>Small dataset</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-50">50</xref>]</td>
<td>MARBERT and BERT</td>
<td>Arabic</td>
<td>Around 24K tweets</td>
<td>N/A</td>
<td>F1-score 75%</td>
<td>Spam detection</td>
<td>Needs improvement</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-51">51</xref>]</td>
<td>Machine Learning, SVM, NB</td>
<td>Arabic</td>
<td>30K tweets and comments from Twitter/YouTube</td>
<td>TF-IDF BoW</td>
<td>Accuracy SVM: 95.742% NB: 70.942%</td>
<td>Good accuracy</td>
<td>Tweets mixed with comments</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-52">52</xref>]</td>
<td>LSTM Bi-LSTM</td>
<td>Arabic</td>
<td>10K tweets from Twitter</td>
<td>N/A</td>
<td>Accuracy 88% for LSTM and BiLSTM</td>
<td>Three datasets used for analysis</td>
<td>Short and imbalance dataset</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-53">53</xref>]</td>
<td>Machine Learning</td>
<td>Arabic</td>
<td>Twitter with 4140 tweets</td>
<td>AraBERT, TF-IDF</td>
<td>Accuracy 89%</td>
<td>Three datasets used</td>
<td>Limited instances</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Methodology</title>
<p>This section covers the steps involved in the proposed study that can be considered as the theoretical framework of the study: Dataset collection, dataset preprocessing, features extraction, generating the models, and evaluation metrics. Different phases of the proposed methodology are shown in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Different phases of the proposed methodology</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48003-fig-2.tif"/>
</fig>
<sec id="s4_1">
<label>4.1</label>
<title>Dataset Collection and Labeling</title>
<p>The dataset was collected from Arabic Twitter using the API, and it consists of about 9K tweets from diverse users from the Middle East, especially Saudi Arabia. It provides a linguistic variety in terms of diverse dialects used in the Arab world, not just one standard dialect such as modern standard Arabic (MSA), that is mainly used in similar studies conducted in Saudi Arabia. The reason behind the MSA is that most of the urban areas Arabic speaking users utilize MSA during their conversations [<xref ref-type="bibr" rid="ref-50">50</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>]. While in the rural and semi-urban areas, nonstandard, slang and local diverse dialects are mainly observed during the data collection process. The collected tweets dataset was manually labeled. In this regard, it was divided into two labels or classes: Bullying and non-bullying for binary classification. Within the Bullying class, tweets contained at least one bullying word (considering the standard Arabic dictionary), and in the non-bullying class, tweets are entirely free of them.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Dataset Preprocessing</title>
<p>Preprocessing is a critical stage in a machine learning model since it cleans and prepares the dataset so that it may be used to train the classifier. In our case, the tweets are written in various dialects rather than traditional Arabic. Therefore, we have applied NLP approaches to cope with various issues posed by tweets in the Arabic language. It was applied as follows:</p>
<sec id="s4_2_1">
<label>4.2.1</label>
<title>Data Cleaning, Handling Missing Values and Outliers</title>
<p>Before any work is done on the dataset, it needs to be cleaned from the inconsistent data by deleting the duplicate and broken (inconsistent) Tweets. The missing values and outliers were handled by the standard NLP preprocessing methods such as imputation [<xref ref-type="bibr" rid="ref-8">8</xref>]. Oversampling technique has been used to balance the collected dataset [<xref ref-type="bibr" rid="ref-8">8</xref>]. Nonetheless, no outliers have been detected in the dataset.</p>
</sec>
<sec id="s4_2_2">
<label>4.2.2</label>
<title>Normalization</title>
<p>The dataset was cleaned and normalized into a uniform text. This process was implemented by Python programming language using &#x201C;pyarabic&#x201D; libraries and regular expressions. Here we remove the Hashtags, Usernames, Numbers, URLs, English Letters, Special Characters, Quotations Marks, Brackets, Emojis, Repeated Letters, Tashkeel (like: &#x201C;<inline-graphic xlink:href="CMC_48003-inline-29.tif"/>&#x201D; to &#x201C;<inline-graphic xlink:href="CMC_48003-inline-30.tif"/>&#x201D;), and Tatweel (like: &#x201C;<inline-graphic xlink:href="CMC_48003-inline-31.tif"/>&#x201D; to &#x201C;<inline-graphic xlink:href="CMC_48003-inline-32.tif"/>&#x201D;) [<xref ref-type="bibr" rid="ref-8">8</xref>].</p>
</sec>
<sec id="s4_2_3">
<label>4.2.3</label>
<title>Stop-Word Removal</title>
<p>Stop-words are meaningless terms that do not aid in the analysis. We developed a stop-word dictionary by collecting it from &#x201C;<ext-link ext-link-type="uri" xlink:href="http://countwordsfree.com">countwordsfree.com</ext-link>&#x201D; (accessed on 10/05/2024). Example stop-words: &#x201C;<inline-graphic xlink:href="CMC_48003-inline-33.tif"/>&#x201D;, &#x201C;<inline-graphic xlink:href="CMC_48003-inline-34.tif"/>&#x201D;, &#x201C;<inline-graphic xlink:href="CMC_48003-inline-35.tif"/>&#x201D;, &#x201C;<inline-graphic xlink:href="CMC_48003-inline-36.tif"/>&#x201D;, etc.</p>
</sec>
<sec id="s4_2_4">
<label>4.2.4</label>
<title>Tokenization</title>
<p>When punctuation and white space marks are encountered, Tokenization is used to split the sentences into tokens. For example, &#x201C;<inline-graphic xlink:href="CMC_48003-inline-37.tif"/>&#x201D; tokenized into &#x201C;<inline-graphic xlink:href="CMC_48003-inline-38.tif"/>&#x201D;, &#x201C;<inline-graphic xlink:href="CMC_48003-inline-39.tif"/>&#x201D;, &#x201C;<inline-graphic xlink:href="CMC_48003-inline-40.tif"/>&#x201D;. The tokenize class from the &#x201C;pyarabic.araby&#x201D; package was used.</p>
</sec>
<sec id="s4_2_5">
<label>4.2.5</label>
<title>Stemming</title>
<p>Stemming is a technique for returning a word to its root by removing suffixes and prefixes. For example: &#x201C;<inline-graphic xlink:href="CMC_48003-inline-41.tif"/>&#x201D; converted into &#x201C;<inline-graphic xlink:href="CMC_48003-inline-42.tif"/>&#x201D;. ISRI Stemmer was utilized in this study and was imported from &#x201C;nltk.stem.isri&#x201D; library.</p>
</sec>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Features Extraction</title>
<p>Focusing on finding the best ML techniques to detect cyberbullying within Arabic tweets, we tested the best model features that gave the highest performance accuracy. N-gram extracted the required features with unigram range and TF-IDF methods. Also, nine supervised ML models were used to train the dataset, including SVM, NB, RF, LR, LightGBM, CatBoost, XGBoost, AdaBoost, and Bagging.</p>
<sec id="s4_3_1">
<label>4.3.1</label>
<title>Term Frequency&#x2013;Inverse Document Frequency (TF-IDF)</title>
<p>TF-IDF is most used for text classification and feature extraction. Term frequency is the number of times a word appears within a document, as given in <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref>. The Inverse Document Frequency returns how common or rare a word is in the entire record set, as given in <xref ref-type="disp-formula" rid="eqn-2">Eq. (2)</xref>. So, if the word is ubiquitous and appears in many records, this number will be 0. Otherwise, it will be 1, as given in <xref ref-type="disp-formula" rid="eqn-3">Eq. (3)</xref>.
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mrow><mml:mtext>TF</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>d</mml:mtext></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mrow><mml:mtext>freq</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>d</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mrow><mml:mtext>IDF</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>t</mml:mtext></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mtext>n</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>DF</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>t</mml:mtext></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:math></disp-formula>
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:mtext>TF</mml:mtext></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mtext>IDF</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>t</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mtext>d</mml:mtext></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mtext>&#xA0;TF</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>t</mml:mtext></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mtext>d</mml:mtext></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mrow><mml:mtext>IDF</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>t</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
</sec>
<sec id="s4_3_2">
<label>4.3.2</label>
<title>Sequence of N Words (N-Gram)</title>
<p>N-grams extract the sequence of N words. In our case, we used Unigram with range (min: 1, max: 1). For example, the sentence &#x201C;<inline-graphic xlink:href="CMC_48003-inline-43.tif"/>&#x201D; should be divided into {&#x2018;<inline-graphic xlink:href="CMC_48003-inline-44.tif"/>&#x2019;, &#x2018;<inline-graphic xlink:href="CMC_48003-inline-45.tif"/>&#x2019;, &#x2018;<inline-graphic xlink:href="CMC_48003-inline-46.tif"/>&#x2019;}.</p>
</sec>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Model Generation and Evaluation</title>
<p>This study implemented nine supervisor machine learning models using Python, Sklearn, Catboost, LightGPM, and XGBoost libraries to determine the best classification model. The accuracy as given in <xref ref-type="disp-formula" rid="eqn-4">Eq. (4)</xref>, precision as given in <xref ref-type="disp-formula" rid="eqn-5">Eq. (5)</xref>, recall as given in <xref ref-type="disp-formula" rid="eqn-6">Eq. (6)</xref>, and F1-score as given in <xref ref-type="disp-formula" rid="eqn-7">Eq. (7)</xref> will be used to evaluate the classifiers in this research. True Positive (TP), True Negative (TN), False Positive (FP), and False Negative (FN) will be determined using these equations [<xref ref-type="bibr" rid="ref-49">49</xref>].
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mrow><mml:mtext>Accuracy</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>TN</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>TN</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>&#xA0;FP</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>FN</mml:mtext></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mrow><mml:mtext>Precision</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>FP</mml:mtext></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mrow><mml:mtext>Recall</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>FN</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mrow><mml:mtext>F1-score</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x2217;</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mtext>precision</mml:mtext></mml:mrow><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mtext>recall</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext>precision</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>recall</mml:mtext></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Results and Discussion</title>
<p>This section presents the results of all classifiers in detail and demonstrates various effects of Unigram and TF-IDF on text classification for each model. After preprocessing the dataset and extracting the features, the dataset will be fed to the classifiers to determine whether a tweet is cyberbullying. To better evaluate the proposed models, two sets of experiments were conducted in this study, the first with the collected dataset and the second with an existing dataset. It is worth noting that no overfitting was observed during the analyses. The following sections summarize the results of the two experiments.</p>
<sec id="s5_1">
<label>5.1</label>
<title>Experimental Results with the Collected Dataset</title>
<p>The models were built using a diverse dataset collected and curated in this experiment. The dataset was collected through Twitter API, consisting of about 9K tweets. Nine classifiers were used in these experiments, which are: SVM, RF, LR, AdaBoost, CatBoost, LightGBM, Bagging, XGBoost, and NB. The outcomes of each classifier&#x2019;s model evaluation predictions are shown in <xref ref-type="table" rid="table-4">Table 4</xref>.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Experimental results with the collected dataset</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Classifier</th>
<th>Feature extraction</th>
<th>Accuracy</th>
<th>Recall</th>
<th>Precision</th>
<th>F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td>SVM</td>
<td>TF-IDF</td>
<td>87.93</td>
<td>83.33</td>
<td>85.57</td>
<td>84.44</td>
</tr>
<tr>
<td/>
<td>Unigram</td>
<td>86.41</td>
<td>83.51</td>
<td>82.18</td>
<td>82.84</td>
</tr>
<tr>
<td>RF</td>
<td>TF-IDF</td>
<td>89.17</td>
<td>80.99</td>
<td><bold>90.44</bold></td>
<td>85.46</td>
</tr>
<tr>
<td/>
<td>Unigram</td>
<td>89.13</td>
<td>82.34</td>
<td>89.17</td>
<td>85.62</td>
</tr>
<tr>
<td>NB</td>
<td>TF-IDF</td>
<td>83.43</td>
<td>75.14</td>
<td>81.29</td>
<td>78.09</td>
</tr>
<tr>
<td/>
<td>Unigram</td>
<td>83.43</td>
<td>75.14</td>
<td>81.29</td>
<td>78.09</td>
</tr>
<tr>
<td>LR</td>
<td>TF-IDF</td>
<td>88.14</td>
<td>78.92</td>
<td>89.66</td>
<td>83.95</td>
</tr>
<tr>
<td/>
<td>Unigram</td>
<td>87.93</td>
<td>82.07</td>
<td>86.51</td>
<td>84.23</td>
</tr>
<tr>
<td>CatBoost</td>
<td>TF-IDF</td>
<td>89.84</td>
<td>84.05</td>
<td>89.45</td>
<td>86.67</td>
</tr>
<tr>
<td/>
<td>Unigram</td>
<td>89.06</td>
<td>82.97</td>
<td>88.47</td>
<td>85.63</td>
</tr>
<tr>
<td>LightGBM</td>
<td>TF-IDF</td>
<td>89.63</td>
<td>84.41</td>
<td>88.65</td>
<td>86.45</td>
</tr>
<tr>
<td/>
<td>Unigram</td>
<td>89.49</td>
<td>83.24</td>
<td>89.28</td>
<td>86.15</td>
</tr>
<tr>
<td>AdaBoost</td>
<td>TF-IDF</td>
<td>88.21</td>
<td>81.44</td>
<td>87.68</td>
<td>84.45</td>
</tr>
<tr>
<td/>
<td>Unigram</td>
<td>88.42</td>
<td>80.9</td>
<td>88.65</td>
<td>84.6</td>
</tr>
<tr>
<td>XGBoost</td>
<td>TF-IDF</td>
<td><bold>89.95</bold></td>
<td>84.23</td>
<td>89.56</td>
<td><bold>88.82</bold></td>
</tr>
<tr>
<td/>
<td>Unigram</td>
<td>89.17</td>
<td>82.7</td>
<td>88.95</td>
<td>85.71</td>
</tr>
<tr>
<td>Bagging</td>
<td>TF-IDF</td>
<td>89.27</td>
<td>86.67</td>
<td>86.12</td>
<td>86.39</td>
</tr>
<tr>
<td/>
<td>Unigram</td>
<td>88.39</td>
<td><bold>88.02</bold></td>
<td>83.36</td>
<td>85.63</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="table" rid="table-4">Table 4</xref> shows that XGBoost has the highest accuracy and F1-score, as it reaches 89.95% and 86.82%, respectively, using TF-IDF as a feature extraction approach. Also, RF achieved 90.44% precision using TF-IDF, considered the highest score in the table. The highest score for the Recall goes for the Bagging classifier, as it obtained 88.02% using Unigram as feature extraction. The TF-IDF approach enhances the classifier&#x2019;s accuracy more than the N-grams, as shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>. To achieve better performance, it is best to use the TF-IDF rather than N-gram for extracting the features from text.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>The impact of feature extraction on text classification</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48003-fig-3.tif"/>
</fig>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Experimental Results with an Existing Dataset</title>
<p>To compare the proposed models with existing research, we investigated the proposed methodology using the dataset used by [<xref ref-type="bibr" rid="ref-33">33</xref>]. This dataset contains 15 K tweets; 39% are labeled as cyberbullying. Their work passed the dataset through preprocessing steps: Tokenization, filtering, normalization, and stemming and then they used N-gram and SVM. In the proposed work, we improved accuracy as follows. First, the dataset was imbalanced, so we added more tweets to make it a balanced dataset. Second, the dataset goes through many steps of preprocessing techniques: removing repeated tweets, tokenization, removing noise (numbers, English letters, emojis), normalization, removing stop words, and stemming. After that, we applied TF-IDF and N-gram for feature extraction and SVM for classification, following the same steps. It is apparent from <xref ref-type="table" rid="table-4">Table 4</xref> that the proposed approach achieves higher values of evaluation metrics than [<xref ref-type="bibr" rid="ref-33">33</xref>] with the same classifier. Furthermore, we applied LR, to investigate the classification result. TF-IDF with SVM has the highest results, followed by N-gram with LR and N-gram with SVM. The comparison is given in <xref ref-type="table" rid="table-5">Table 5</xref>. It is also apparent that the proposed algorithms are also promising in terms of scalability. The possible reason behind these outcomes is TF-IDF that have proven to be instrumental in NLP overall. On contrary, SVM has been investigated with N-gram, but it did not provide a similar performance. The same is evident in <xref ref-type="table" rid="table-6">Table 6</xref> where TF-IDF has been combined with XGBoost. As far as the computational complexity of the proposed models is concerned, during training ensemble models take longer than the traditional models in terms of convergence rate. It is obvious due to the nature of the ensemble algorithms where the consensus takes more time.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Comparison with an existing dataset</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Approach</th>
<th>Feature extraction</th>
<th>Classifier</th>
<th>Accuracy</th>
<th>Precision</th>
<th>Recall</th>
<th>F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td>Alakrot et al. (2018) [<xref ref-type="bibr" rid="ref-33">33</xref>]</td>
<td>N-gram</td>
<td>SVM</td>
<td>85%</td>
<td>81%</td>
<td>78%</td>
<td>80%</td>
</tr>
<tr>
<td rowspan="3">Proposed approach</td>
<td>N-gram</td>
<td>SVM</td>
<td>86.01%</td>
<td>88.67%</td>
<td>82.27%</td>
<td>85.35%</td>
</tr>
<tr>
<td>TF-IDF</td>
<td>SVM</td>
<td>86.55%</td>
<td>90.20%</td>
<td>81.74%</td>
<td>85.76%</td>
</tr>
<tr>
<td>N-gram</td>
<td>LR</td>
<td>86.42%</td>
<td>89.27%</td>
<td>82.5%</td>
<td>85.75%</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>Comparison with state-of-the-art</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Approach</th>
<th>Feature extraction</th>
<th>Classifier</th>
<th>Accuracy</th>
<th>Precision</th>
<th>Recall</th>
<th>F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="2">Alduailaj et al. (2023) [<xref ref-type="bibr" rid="ref-51">51</xref>]</td>
<td>TF-IDF</td>
<td>SVM</td>
<td>95.742%</td>
<td>92%</td>
<td>84%</td>
<td>88%</td>
</tr>
<tr>
<td>BoW</td>
<td>NB</td>
<td>70.942%</td>
<td></td>
<td></td>
<td></td>
</tr>
<tr>
<td>Alzaqebah et al. (2023) [<xref ref-type="bibr" rid="ref-52">52</xref>]</td>
<td>None</td>
<td>LSTM</td>
<td>88%</td>
<td>88%</td>
<td>88%</td>
<td>88%</td>
</tr>
<tr>
<td>Mursi et al. (2023) [<xref ref-type="bibr" rid="ref-53">53</xref>]</td>
<td>TF-IDF</td>
<td>MLP</td>
<td>89%</td>
<td>88%</td>
<td>90%</td>
<td>89%</td>
</tr>
<tr>
<td rowspan="3"><bold>Proposed</bold></td>
<td>TF-IDF</td>
<td>XGBoost</td>
<td rowspan="3">89.95%</td>
<td rowspan="3">90.44%</td>
<td rowspan="3">88.02%</td>
<td rowspan="3">88.82%</td>
</tr>
<tr>
<td>N-gram</td>
<td>RF</td>
</tr>
<tr>
<td>Unigram</td>
<td>Bagging</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5_3">
<label>5.3</label>
<title>Qualitative Comparison with State-of-the-Art</title>
<p>To compare the scheme with the state-of-the-art, we have selected some recent studies with three common aspects (Language (Arabic), social media (Twitter), and Data collection region). However, their dataset was different than the one proposed. In this regard, a comparison was made with three recent approaches [<xref ref-type="bibr" rid="ref-52">52</xref>&#x2013;<xref ref-type="bibr" rid="ref-54">54</xref>] from 2023. The proposed scheme outperforms the schemes given in [<xref ref-type="bibr" rid="ref-52">52</xref>] with NB, [<xref ref-type="bibr" rid="ref-53">53</xref>], and [<xref ref-type="bibr" rid="ref-54">54</xref>] in terms of accuracy by 19%, 1.95%, and 0.95%, respectively, while using XGBoost with TF-IDF. However, the scheme in [<xref ref-type="bibr" rid="ref-52">52</xref>] outperformed the proposed scheme regarding accuracy and precision using the SVM classifier. Nonetheless, the authors mentioned in [<xref ref-type="bibr" rid="ref-52">52</xref>] that the results are with a segmented scenario and not generalized for their whole dataset. In terms of precision, the proposed scheme using RF with N-gram performs better than [<xref ref-type="bibr" rid="ref-53">53</xref>&#x2013;<xref ref-type="bibr" rid="ref-54">54</xref>]. Regarding Recall and F1-score, the proposed scheme is better than [<xref ref-type="bibr" rid="ref-53">53</xref>] with a litter margin. Nonetheless, the scheme in [<xref ref-type="bibr" rid="ref-54">54</xref>] marginally performs better than the proposed scheme regarding Recall and F1-score. The comparison with state-of-the-art is presented in <xref ref-type="table" rid="table-6">Table 6</xref>.</p>

</sec>
<sec id="s5_4">
<label>5.4</label>
<title>Discussion</title>
<p>The proposed scheme addresses the cyberbullying detection problem from Arabic tweets. It is from the hottest areas of research in behavioral studies and modelling of online users [<xref ref-type="bibr" rid="ref-54">54</xref>]. The dataset in this regard has been collected from diverse Arab regions, annotated, and preprocessed before applying a broad spectrum of approaches in machine learning. The analyses are made with the same dataset and a secondary dataset from the literature. The proposed scheme was promising in both cases. Wholistically, it was observed that in terms of feature extraction methods, the schemes involving TF-IDF performed better than other feature selection methods such as BoW, and N-gram. Moreover, the preprocessing techniques used in the studies with the diverse dialects used in the Arabic language tweets make a difference. In contrast to schemes in [<xref ref-type="bibr" rid="ref-33">33</xref>,<xref ref-type="bibr" rid="ref-52">52</xref>&#x2013;<xref ref-type="bibr" rid="ref-54">54</xref>], the proposed scheme is comparable in all metrics. The dataset was collected by means of Twitter APIs and fused prior to investigation, and it does not contain any users&#x2019; personal information of any type. The scheme can easily be generalized and/or extended to other social media platforms such as YouTube and Facebook comments. The findings of the study can be interpreted as: the models can be used in social media platforms to combat cyberbullying by identifying potential users and contents containing such offensive language and taking strict actions like reporting and blocking the accounts to prevent the emotional and phycological damage to the victims of cyberbullying.</p>
</sec>
<sec id="s5_5">
<label>5.5</label>
<title>Limitations of the Study and Future Work</title>
<p>As far as the limitation of the study is concerned, the dataset is limited in terms of the number of instances, though it contains diversified tweet samples collected from diverse middle eastern regions with various dialects. Those regions include Iraq, UAE, Oman, Kuwait, Qatar, Jordan, and other Arabic speaking nations. The scheme is robust against the dialectic variations and capable of handling them effectively as evident in the results and discussion section. Nonetheless, the major chunk of dataset was collected from Saudi Arabia&#x2019;s major cities like Riyadh, Dammam, and Jeddah where modern standard Arabic (MSA) dialect is usually evident in social media. That could be a potential bias in the dataset. Such issues can be overcome by using data augmentation techniques by adding more data with a uniform distribution and balanced sampling techniques. Moreover, accuracy and other metrics can be improved by using advanced data preprocessing techniques with different feature extraction methods [<xref ref-type="bibr" rid="ref-55">55</xref>], deep learning and ensemble learning approaches. Additionally, the encoders/transformers like Bidirectional Encoder Representations from Transformers (BERT) with their Arabic counter parts namely ARBERT (Arabic BERT) and MARBERT (Modern Standard Arabic with BERT) can also be investigated. MARBERT is a large-scale pre-trained masked language model focused on both Dialectal Arabic (DA) and MSA [<xref ref-type="bibr" rid="ref-50">50</xref>].</p>
</sec>
</sec>
<sec id="s6">
<label>6</label>
<title>Conclusions</title>
<p>Detection of cyberbullying is getting more difficult as internet users have too many ways of bullying without being identified. Cyberbullying can threaten individuals and cause the victims to commit suicide or go into depression, so its detection is necessary. Several studies have been conducted in the literature but mainly in English, while only a few studies exist in Arabic. In this study, we have proposed and developed Machine Learning models for Cyberbullying Detection from Arabic tweets with diverse dialects. We improved the proposed model significantly using feature extraction methods. The dataset achieved high results using XGBoost, Bagging, and RF classifiers, with XGBoost getting the highest accuracy of them all at 89.95%, using TF-IDF as feature extraction. In the future, we intend to enhance the dataset by data augmentation techniques. Moreover, in future, we aim to improve the accuracy of the detection method by applying hybrid models, investigating diverse datasets, and more than two feature extraction methods for the NLP, which will help further fine-tune the models.</p>
</sec>
</body>
<back>
<ack>
<p>The authors like to acknowledge CCSIT, IAU, Dammam for using the resources.</p>
</ack>
<sec><title>Funding Statement</title>
<p>The authors received no specific funding for this study.</p>
</sec>
<sec><title>Author Contributions</title>
<p>Conceptualization, Dhiaa Musleh; Data curation, Minhal AlBo-Hassan, Jawad Alnemer and Ghazy Al-Mutairi; Formal analysis, Atta Rahman, Hayder Alsebaa and Dalal Aldowaihi; Funding acquisition, May Aldossary and Fahd Alhaidari; Investigation, Dalal Aldowaihi and Fahd Alhaidari; Methodology, Dhiaa Musleh, Mohammed Alkherallah and Hayder Alsebaa; Project administration, Atta Rahman; Resources, Jawad Alnemer; Software, Mohammed Alkherallah, Minhal Al-Bo-Hassan, Mustafa Alawami and Jawad Alnemer; Supervision, Dhiaa Musleh and Dalal Aldo-waihi; Validation, Mustafa Alawami, Hayder Alsebaa and May Aldossary; Visualization, Ghazy Al-Mutairi; Writing&#x2013;original draft, Mohammed Alkherallah, Minhal AlBo-Hassan, Mustafa Alawami and Ghazy Al-Mutairi; Writing&#x2013;review &#x0026; editing, Atta Rahman, May Aldossary and Fahd Alhaidari. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability"><title>Availability of Data and Materials</title>
<p>The data can be requested from the corresponding author.</p>
</sec>
<sec sec-type="COI-statement"><title>Conflicts of Interest</title>
<p>The authors declare that they have no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>W. N.</given-names> <surname>Hamiza Wan Ali</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Mohd</surname></string-name>, and <string-name><given-names>F.</given-names> <surname>Fauzi</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection: An overview</article-title>,&#x201D; in <conf-name>2018 Cyber Resilience Conf.</conf-name>, <publisher-loc>Putrajaya, Malaysia</publisher-loc>, <year>2018</year>, pp. <fpage>1</fpage>&#x2013;<lpage>3</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CR.2018.8626869</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Sri Nandhini</surname></string-name> and <string-name><given-names>J. I.</given-names> <surname>Sheeba</surname></string-name></person-group>, &#x201C;<article-title>Online social network bullying detection using intelligence technique</article-title>,&#x201D; <source>Procedia Comput. Sci.</source>, vol. <volume>45</volume>, no. <issue>1</issue>, pp. <fpage>485</fpage>&#x2013;<lpage>492</lpage>, <year>2015</year>. doi: <pub-id pub-id-type="doi">10.1016/j.procs.2015.03.085</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Iqbal</surname></string-name></person-group>, &#x201C;<article-title>Twitter revenue and usage statistics (2021) Business of App</article-title>,&#x201D; <month>Jul</month>. <day>5</day>, <year>2021</year>. Accessed: Sep. 14, 2021. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.businessofapps.com/data/twitter-statistics/">https://www.businessofapps.com/data/twitter-statistics/</ext-link></mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Label</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying statistics</article-title>,&#x201D; Accessed: Jan. 16, 2021. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.ditchthelabel.org/cyber-bullying-statistics-what-they-tell-us">https://www.ditchthelabel.org/cyber-bullying-statistics-what-they-tell-us</ext-link></mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="other">&#x201C;<article-title>Most common languages used on the internet as of January 2020, by share of internet users</article-title>,&#x201D; <month>Jun</month>. <year>2020</year>. Accessed: Sep. 14, 2021. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.statista.com/statistics/262946/share-of-the-most-common-languages-on-the-internet/">https://www.statista.com/statistics/262946/share-of-the-most-common-languages-on-the-internet/</ext-link></mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Patchin</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Hinduja</surname></string-name></person-group>, &#x201C;<article-title>Measuring cyberbullying: Implications for research</article-title>,&#x201D; <source>Aggress. Violent Behav.</source>, vol. <volume>23</volume>, no. <issue>4</issue>, pp. <fpage>69</fpage>&#x2013;<lpage>74</lpage>, <month>Jul.&#x2013;Aug</month>. <year>2015</year>. doi: <pub-id pub-id-type="doi">10.1016/j.avb.2015.05.013</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D. N.</given-names> <surname>Hoff</surname></string-name> and <string-name><given-names>S. N.</given-names> <surname>Mitchell</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying: Causes, effects, and remedies</article-title>,&#x201D; <source>J. Educ. Adm.</source>, vol. <volume>47</volume>, no. <issue>5</issue>, pp. <fpage>652</fpage>&#x2013;<lpage>665</lpage>, <month>Aug</month>. <year>2009</year>. doi: <pub-id pub-id-type="doi">10.1108/09578230910981107</pub-id>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Alqarni</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Rahman</surname></string-name></person-group>, &#x201C;<article-title>Arabic tweets-based sentiment analysis to investigate the impact of COVID-19 in KSA: A deep learning approach</article-title>,&#x201D; <source>Big Data Cogn. Comput.</source>, vol. <volume>7</volume>, no. <issue>1</issue>, pp. <fpage>16</fpage>, <year>2023</year>. doi: <pub-id pub-id-type="doi">10.3390/bdcc7010016</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>W. J.</given-names> <surname>Hutchins</surname></string-name></person-group>, &#x201C;<article-title>The georgetown-IBM experiment demonstrated in January 1954</article-title>,&#x201D; in <conf-name>6th Conf. Assoc. Mach. Translat. Americas</conf-name>, <publisher-loc>Washington DC, USA</publisher-loc>, <year>2004</year>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>S. A.</given-names> <surname>Mandal</surname></string-name></person-group>, &#x201C;<article-title>Evolution of machine translation</article-title>,&#x201D; <source>Towards Data Science</source>, Accessed: Jan. 4, 2023. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://towardsdatascience.com/evolution-of-machine-translation-5524f1c88b25">https://towardsdatascience.com/evolution-of-machine-translation-5524f1c88b25</ext-link></mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Almutiry</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Abdel Fattah</surname></string-name></person-group>, &#x201C;<article-title>Arabic cyberbullying detection using Arabic sentiment</article-title>,&#x201D; <source>Egyptian J. Lang. Eng.</source>, vol. <volume>8</volume>, no. <issue>1</issue>, pp. <fpage>39</fpage>&#x2013;<lpage>50</lpage>, <month>Apr</month>. <year>2021</year>. doi: <pub-id pub-id-type="doi">10.21608/ejle.2021.50240.1017</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Kanan</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Aldaaja</surname></string-name>, and <string-name><given-names>B.</given-names> <surname>Hawashin</surname></string-name></person-group>, &#x201C;<article-title>Cyber-bullying and cyber-harassment detection using supervised machine learning techniques in Arabic social media contents</article-title>,&#x201D; <source>J. Int. Technol.</source>, vol. <volume>21</volume>, no. <issue>5</issue>, pp. <fpage>1409</fpage>&#x2013;<lpage>1421</lpage>, <month>Sep</month>. <year>2020</year>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>I.</given-names> <surname>Abu El-Khair</surname></string-name></person-group>, &#x201C;<article-title>Effects of stop words elimination on Arabic information retrieval</article-title>,&#x201D; <source>Int. J. Comput. &#x0026; Inform. Sci.</source>, vol. <volume>4</volume>, no. <issue>3</issue>, pp. <fpage>110</fpage>&#x2013;<lpage>133</lpage>, <year>2006</year>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. L.</given-names> <surname>Aouragh</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Yousfi</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Laaroussi</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Gueddah</surname></string-name>, and <string-name><given-names>M.</given-names> <surname>Nejja</surname></string-name></person-group>, &#x201C;<article-title>A new estimate of the N-gram language model</article-title>,&#x201D; <source>Procedia Comput. Sci.</source>, vol. <volume>189</volume>, no. <issue>1</issue>, pp. <fpage>211</fpage>&#x2013;<lpage>215</lpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.1016/j.procs.2021.05.111</pub-id>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Srinidhi</surname></string-name></person-group>, &#x201C;<article-title>Understanding Word N-grams and N-gram probability in natural language processing</article-title>,&#x201D; <source>Towards Data Science</source>, Accessed: Apr. 23, 2021. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://towardsdatascience.com/understanding-word-n-grams-and-n-gram-probability-in-natural-language-processing-9d9eef0fa058">https://towardsdatascience.com/understanding-word-n-grams-and-n-gram-probability-in-natural-language-processing-9d9eef0fa058</ext-link></mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. A.</given-names> <surname>Al-Ajlan</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Ykhlef</surname></string-name></person-group>, &#x201C;<article-title>Deep learning algorithm for cyberbullying detection</article-title>,&#x201D; <source>Int. J. Adv. Comput. Sci. Appl.</source>, vol. <volume>9</volume>, no. <issue>9</issue>, pp. <fpage>199</fpage>&#x2013;<lpage>205</lpage>, <year>2018</year>. doi: <pub-id pub-id-type="doi">10.14569/IJACSA.2018.090927</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Haidar</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Chamoun</surname></string-name>, and <string-name><given-names>A.</given-names> <surname>Serhrouchni</surname></string-name></person-group>, &#x201C;<article-title>Arabic cyberbullying detection: Using deep learning</article-title>,&#x201D; in <conf-name>2018 7th Int. Conf. Comput. Commun. Eng.</conf-name>, <publisher-loc>Kuala Lumpur, Malaysia</publisher-loc>, <year>2018</year>, pp. <fpage>284</fpage>&#x2013;<lpage>289</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICCCE.2018.8539303</pub-id>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Haidar</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Chamoun</surname></string-name>, and <string-name><given-names>A.</given-names> <surname>Serhrouchni</surname></string-name></person-group>, &#x201C;<article-title>A multilingual system for cyberbullying detection: Arabic content detection using machine learning</article-title>,&#x201D; <source>Adv. Sci. Technol. Eng. Syst. J.</source>, vol. <volume>2</volume>, no. <issue>6</issue>, pp. <fpage>275</fpage>&#x2013;<lpage>284</lpage>, <year>2017</year>. doi: <pub-id pub-id-type="doi">10.25046/aj020634</pub-id>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. S.</given-names> <surname>AlHarbi</surname></string-name>, <string-name><given-names>B. Y.</given-names> <surname>AlHarbi</surname></string-name>, <string-name><given-names>N. J.</given-names> <surname>AlZahrani</surname></string-name>, <string-name><given-names>M. M.</given-names> <surname>Alsheail</surname></string-name>, <string-name><given-names>J. F.</given-names> <surname>Alshobaili</surname></string-name> and <string-name><given-names>D. M.</given-names> <surname>Ibrahim</surname></string-name></person-group>, &#x201C;<article-title>Automatic cyber bullying detection in Arabic social media</article-title>,&#x201D; <source>Int. J. Eng. Res. Technol.</source>, vol. <volume>12</volume>, no. <issue>12</issue>, pp. <fpage>2330</fpage>&#x2013;<lpage>2335</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Mouheb</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Albarghash</surname></string-name>, <string-name><given-names>M. F.</given-names> <surname>Mowakeh</surname></string-name>, <string-name><given-names>Z. A.</given-names> <surname>Aghbari</surname></string-name>, and <string-name><given-names>I.</given-names> <surname>Kamel</surname></string-name></person-group>, &#x201C;<article-title>Detection of Arabic cyberbullying on social networks using machine learning</article-title>,&#x201D; in <conf-name>2019 IEEE/ACS 16th Int. Conf. Comput. Syst. Appl.</conf-name>, <publisher-loc>Abu Dhabi, UAE</publisher-loc>, <year>2019</year>, pp. <fpage>1</fpage>&#x2013;<lpage>5</lpage>. doi: <pub-id pub-id-type="doi">10.1109/AICCSA47632.2019.9035276</pub-id>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Haidar</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Chamoun</surname></string-name>, and <string-name><given-names>A.</given-names> <surname>Serhrouchni</surname></string-name></person-group>, &#x201C;<article-title>Arabic cyberbullying detection: Enhancing performance by using ensemble machine learning</article-title>,&#x201D; in <conf-name>Proc. IEEE Joint ithings, GreenCom, CPSCom) and SmartData</conf-name>, <publisher-loc>Atlanta, GA, USA</publisher-loc>, <year>2019</year>, pp. <fpage>323</fpage>&#x2013;<lpage>327</lpage>. doi: <pub-id pub-id-type="doi">10.1109/iThings/GreenCom/CPSCom/SmartData.2019.00074</pub-id>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Phanomtip</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Sueb-in</surname></string-name>, and <string-name><given-names>S.</given-names> <surname>Vittayakorn</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection on tweets</article-title>,&#x201D; in <conf-name>2021 18th Int. Conf. Elect. Eng./Electron., Comput., Telecommun. Inf. Technol.</conf-name>, <publisher-loc>Chiang Mai, Thailand</publisher-loc>, <year>2021</year>, pp. <fpage>295</fpage>&#x2013;<lpage>298</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ECTI-CON51831.2021.9454848</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>V.</given-names> <surname>Banerjee</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Telavane</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Gaikwad</surname></string-name>, and <string-name><given-names>P.</given-names> <surname>Vartak</surname></string-name></person-group>, &#x201C;<article-title>Detection of cyberbullying using deep neural network</article-title>,&#x201D; in <conf-name>Proc. ICACCS</conf-name>, <publisher-loc>Coimbatore, India</publisher-loc>, <year>2019</year>, pp. <fpage>604</fpage>&#x2013;<lpage>607</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICACCS.2019.8728378</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Bharti</surname></string-name>, <string-name><given-names>A. K.</given-names> <surname>Yadav</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Kumar</surname></string-name>, and <string-name><given-names>D.</given-names> <surname>Yadav</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection from tweets using deep learning</article-title>,&#x201D; <source>Kybernetes</source>, vol. <volume>51</volume>, no. <issue>9</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>13</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Lokhande</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Suryawanshi</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Kaulgud</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Joshi</surname></string-name>, and <string-name><given-names>P.</given-names> <surname>Ingle</surname></string-name></person-group>, &#x201C;<article-title>Detecting cyber bullying on twitter using machine learning techniques</article-title>,&#x201D; <source>J. Critical Rev.</source>, vol. <volume>7</volume>, no. <issue>19</issue>, pp. <fpage>1090</fpage>&#x2013;<lpage>1094</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A. R.</given-names> <surname>Almutairi</surname></string-name> and <string-name><given-names>M. A.</given-names> <surname>Al-Hagery</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection by sentiment analysis of tweets&#x0027; contents written in Arabic in Saudi Arabia society</article-title>,&#x201D; <source>Int. J. Comput. Sci. Netw. Secur.</source>, vol. <volume>21</volume>, no. <issue>3</issue>, pp. <fpage>112</fpage>&#x2013;<lpage>119</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>V.</given-names> <surname>Jain</surname></string-name>, <string-name><given-names>V.</given-names> <surname>Kumar</surname></string-name>, <string-name><given-names>V.</given-names> <surname>Pal</surname></string-name>, and <string-name><given-names>D. K.</given-names> <surname>Vishwakarma</surname></string-name></person-group>, &#x201C;<article-title>Detection of cyberbullying on social media using machine learning</article-title>,&#x201D; in <conf-name>2021 5th Int. Conf. Comput. Methodologies Commun.</conf-name>, <publisher-loc>Erode, India</publisher-loc>, <year>2021</year>, pp. <fpage>1091</fpage>&#x2013;<lpage>1096</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICCMC51019.2021.9418254</pub-id>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Al-Mamun</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Akhter</surname></string-name></person-group>, &#x201C;<article-title>Social media bullying detection using machine learning on Bangla text</article-title>,&#x201D; in <conf-name>2018 10th Int. Conf. Elect. Comput. Eng.</conf-name>, <publisher-loc>Dhaka, Bangladesh</publisher-loc>, <year>2018</year>, pp. <fpage>385</fpage>&#x2013;<lpage>388</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICECE.2018.8636797</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>R. R.</given-names> <surname>Dalvi</surname></string-name>, <string-name><given-names>S. B.</given-names> <surname>Chavan</surname></string-name>, and <string-name><given-names>A.</given-names> <surname>Halbe</surname></string-name></person-group>, &#x201C;<article-title>Detecting twitter cyberbullying using machine learning</article-title>,&#x201D; in <conf-name>2020 4th Int. Conf. Intell. Comput. Control Syst.</conf-name>, <publisher-loc>Madurai, India</publisher-loc>, <month>May</month> <year>2020</year>, pp. <fpage>297</fpage>&#x2013;<lpage>301</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICICCS48265.2020.9120893</pub-id>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>V.</given-names> <surname>Balakrishnan</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Khana</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Fernandez</surname></string-name>, and <string-name><given-names>H. R.</given-names> <surname>Arabnia</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection on twitter using big five and dark triad features</article-title>,&#x201D; <source>Pers. Individ. Dif.</source>, vol. <volume>141</volume>, no. <issue>15</issue>, pp. <fpage>252</fpage>&#x2013;<lpage>257</lpage>, <month>Apr.</month> <year>2019</year>. doi: <pub-id pub-id-type="doi">10.1016/j.paid.2019.01.024</pub-id>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Agrawal</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Awekar</surname></string-name></person-group>, &#x201C;<article-title>Deep learning for detecting cyberbullying across multiple social media platforms</article-title>,&#x201D; in <conf-name>Proc. ECIR 2018, Adv. Inform. Retrieval</conf-name>, <publisher-loc>Grenoble, France</publisher-loc>, <year>2018</year>, pp. <fpage>141</fpage>&#x2013;<lpage>153</lpage>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>di Capua</surname></string-name>, <string-name><given-names>E.</given-names> <surname>di Nardo</surname></string-name>, and <string-name><given-names>A.</given-names> <surname>Petrosino</surname></string-name></person-group>, &#x201C;<article-title>Unsupervised cyber bullying detection in social networks</article-title>,&#x201D; in <conf-name>2016 23rd Int. Conf. Pattern Recognit.</conf-name>, <publisher-loc>Cancun, Mexico</publisher-loc>, <year>2016</year>, pp. <fpage>432</fpage>&#x2013;<lpage>437</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICPR.2016.7899672</pub-id>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Alakrot</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Murray</surname></string-name>, and <string-name><given-names>N.</given-names> <surname>Nikolov</surname></string-name></person-group>, &#x201C;<article-title>Towards accurate detection of offensive language in online communication in Arabic</article-title>,&#x201D; <source>Procedia Comput. Sci.</source>, vol. <volume>142</volume>, no. <issue>1</issue>, pp. <fpage>315</fpage>&#x2013;<lpage>320</lpage>, <year>2018</year>. doi: <pub-id pub-id-type="doi">10.1016/j.procs.2018.10.491</pub-id>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Reynolds</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Kontostathis</surname></string-name>, and <string-name><given-names>L.</given-names> <surname>Edwards</surname></string-name></person-group>, &#x201C;<article-title>Using machine learning to detect cyberbullying</article-title>,&#x201D; in <conf-name>Proc. 10th Int. Conf. ML and Appl. and Workshops</conf-name>, <publisher-loc>Honolulu, HI, USA</publisher-loc>, <year>2011</year>, pp. <fpage>241</fpage>&#x2013;<lpage>244</lpage>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Maral</surname></string-name> and <string-name><given-names>K.</given-names> <surname>Eckert</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection in social networks using deep learning based models; a reproducibility study</article-title>,&#x201D; in <conf-name>2016 23rd Int. Conf. Pattern Recognit.</conf-name>, <publisher-loc>Bratislava, Slovakia</publisher-loc>, <year>2020</year>, pp. <fpage>245</fpage>&#x2013;<lpage>255</lpage>. doi: <pub-id pub-id-type="doi">10.1109/ICPR.2016.7899672</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Hani</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Nashaat</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Ahmed</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Emad</surname></string-name>, <string-name><given-names>E.</given-names> <surname>Amer</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Mohammed</surname></string-name></person-group>, &#x201C;<article-title>Social media cyberbullying detection using machine learning</article-title>,&#x201D; <source>Int. J. Adv. Comput. Sci. Appl.</source>, vol. <volume>10</volume>, no. <issue>5</issue>, pp. <fpage>703</fpage>&#x2013;<lpage>707</lpage>, <year>2019</year>. doi: <pub-id pub-id-type="doi">10.14569/issn.2156-5570</pub-id>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Srivastava</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Khan</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Iwendi</surname></string-name>, and <string-name><given-names>P. K.</given-names> <surname>Reddy Maddikunta</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection solutions based on deep learning</article-title>,&#x201D; <source>Multimed. Syst.</source>, vol. <volume>29</volume>, no. <issue>1</issue>, pp. <fpage>1839</fpage>&#x2013;<lpage>1852</lpage>, <year>2023</year>. doi: <pub-id pub-id-type="doi">10.1007/s00530-020-00701-5</pub-id>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>B. A.</given-names> <surname>Rachid</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Azza</surname></string-name>, and <string-name><given-names>H. H.</given-names> <surname>Ben Ghezala</surname></string-name></person-group>, &#x201C;<article-title>Classification of cyberbullying text in Arabic</article-title>,&#x201D; in <conf-name>2020 Int. Joint Conf. on Neural Netw. (IJCNN)
</conf-name>, <publisher-loc>Glasgow, UK</publisher-loc>, <year>2020</year>, pp. <fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1109/IJCNN48605.2020.9206643</pub-id>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Husain</surname></string-name></person-group>, &#x201C;<article-title>Arabic offensive language detection using machine learning and ensemble</article-title>,&#x201D; <comment>arXiv:2005.08946</comment>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Alfageh</surname></string-name> and <string-name><given-names>T.</given-names> <surname>Alsubait</surname></string-name></person-group>, &#x201C;<article-title>Comparison of machine learning techniques for cyberbullying</article-title>,&#x201D; <source>Int. J. Comput. Sci. Netw. Secur.</source>, vol. <volume>21</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>5</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>I.</given-names> <surname>Nazar</surname></string-name>, <string-name><given-names>D. S.</given-names> <surname>Zois</surname></string-name>, and <string-name><given-names>M.</given-names> <surname>Yao</surname></string-name></person-group>, &#x201C;<article-title>A hierarchical approach for timely cyberbullying detection</article-title>,&#x201D; in <conf-name>2019 IEEE Data Sci. Workshop (DSW)</conf-name>, <publisher-loc>Minneapolis, MN, USA</publisher-loc>, <year>2019</year>, pp. <fpage>190</fpage>&#x2013;<lpage>195</lpage>. doi: <pub-id pub-id-type="doi">10.1109/DSW.2019.8755598</pub-id>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Yin</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Xue</surname></string-name>, and <string-name><given-names>L.</given-names> <surname>Hong</surname></string-name></person-group>, &#x201C;<article-title>Detection of harassment on Web 2.0</article-title>,&#x201D; in <conf-name>Proc. Content Anal. Web Workshop at Www</conf-name>, <publisher-loc>Madrid, Spain</publisher-loc>, <month>Apr</month>. <day>21</day>, <year>2009</year>, pp. <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>N.</given-names> <surname>Ejaz</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Razi</surname></string-name>, and <string-name><given-names>S.</given-names> <surname>Choudhury</surname></string-name></person-group>, &#x201C;<article-title>Towards comprehensive cyberbullying detection: A dataset incorporating aggressive texts, repetition, peerness, and intent to harm</article-title>,&#x201D; <source>Comput. Human Behav.</source>, vol. <volume>153</volume>, no. <issue>3</issue>, pp. <fpage>108123</fpage>, <year>2024</year>. doi: <pub-id pub-id-type="doi">10.1016/j.chb.2023.108123</pub-id>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>S. A.</given-names> <surname>&#x00D6;zel</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Akdemir</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Aksu</surname></string-name>, and <string-name><given-names>E.</given-names> <surname>Sara&#x00E7;</surname></string-name></person-group>, &#x201C;<article-title>Detection of cyberbullying on social media messages in Turkish</article-title>,&#x201D; in <conf-name>2017 Int. Conf. Comput. Sci. Eng. (UBMK)</conf-name>, <publisher-loc>Antalya, Turkey</publisher-loc>, <year>2017</year>, pp. <fpage>366</fpage>&#x2013;<lpage>370</lpage>. doi: <pub-id pub-id-type="doi">10.1109/UBMK.2017.8093411</pub-id>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Dadvar</surname></string-name> and <string-name><given-names>F.</given-names> <surname>deJong</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection: A step toward a safer internet yard</article-title>,&#x201D; in <conf-name>WWW &#x2032;12 Companion: Proc. 21st Int. Conf. on World Wide Web</conf-name>, <publisher-loc>Lyon France</publisher-loc>, <month>Apr.</month> <year>2012</year>, pp. <fpage>121</fpage>. doi: <pub-id pub-id-type="doi">10.1145/2187980.2187995</pub-id>.</mixed-citation></ref>
<ref id="ref-46"><label>[46]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>T. A.</given-names> <surname>Buan</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Steria</surname></string-name>, and <string-name><given-names>R.</given-names> <surname>Ramachandra</surname></string-name></person-group>, &#x201C;<article-title>Automated cyberbullying detection in social media using an SVM activated stacked convolution LSTM network</article-title>,&#x201D; in <conf-name>ICCDA '20: Proc. 2020 4th International Conf. Comput. Data Anal.</conf-name>, <publisher-loc>CA, USA</publisher-loc>, <month>Mar.</month> <year>2020</year>, pp. <fpage>170</fpage>&#x2013;<lpage>174</lpage>.</mixed-citation></ref>
<ref id="ref-47"><label>[47]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>N.</given-names> <surname>Potha</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Maragoudakis</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection using time series modeling</article-title>,&#x201D; in <conf-name>Proc. IEEE Int. Conf. Data Mining Workshop</conf-name>, <publisher-loc>Shenzhen, China</publisher-loc>, <year>2014</year>, pp. <fpage>373</fpage>&#x2013;<lpage>382</lpage>.</mixed-citation></ref>
<ref id="ref-48"><label>[48]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>J. O.</given-names> <surname>Atoum</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection through sentiment analysis</article-title>,&#x201D; in <conf-name>Proc. Int. Conf. Comput. Sci. Comput. Intell. (CSCI)</conf-name>, <publisher-loc>Las Vegas, NV, USA</publisher-loc>, <year>2020</year>, pp. <fpage>292</fpage>&#x2013;<lpage>297</lpage>.</mixed-citation></ref>
<ref id="ref-49"><label>[49]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D. A.</given-names> <surname>Musleh</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>Arabic sentiment analysis of YouTube comments: NLP-Based machine learning approaches for content evaluation</article-title>,&#x201D; <source>Big Data Cogn. Comput.</source>, vol. <volume>7</volume>, no. <issue>127</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>15</lpage>, <year>2023</year>. doi: <pub-id pub-id-type="doi">10.3390/bdcc7030127</pub-id>.</mixed-citation></ref>
<ref id="ref-50"><label>[50]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Alotaibi</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>Spam and sentiment detection in Arabic tweets using MarBert model</article-title>,&#x201D; <source>Math. Model. Eng. Prob.</source>, vol. <volume>9</volume>, no. <issue>6</issue>, pp. <fpage>1574</fpage>&#x2013;<lpage>1582</lpage>, <year>2022</year>. doi: <pub-id pub-id-type="doi">10.18280/mmep.090617</pub-id>.</mixed-citation></ref>
<ref id="ref-51"><label>[51]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A. M.</given-names> <surname>Alduailaj</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Belghith</surname></string-name></person-group>, &#x201C;<article-title>Detecting Arabic cyberbullying tweets using machine learning</article-title>,&#x201D; <source>Mach Learn. Knowl. Extr.</source>, vol. <volume>5</volume>, no. <issue>1</issue>, pp. <fpage>29</fpage>&#x2013;<lpage>42</lpage>, <year>2023</year>. doi: <pub-id pub-id-type="doi">10.3390/make5010003</pub-id>.</mixed-citation></ref>
<ref id="ref-52"><label>[52]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Alzaqebah</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>Cyberbullying detection framework for short and imbalanced Arabic datasets</article-title>,&#x201D; <source>J. King Saud Univ.&#x2013; Comput. Inform. Sci.</source>, vol. <volume>35</volume>, no. <issue>8</issue>, pp. <fpage>101652</fpage>, <year>2023</year>. doi: <pub-id pub-id-type="doi">10.1016/j.jksuci.2023.101652</pub-id>.</mixed-citation></ref>
<ref id="ref-53"><label>[53]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K. T.</given-names> <surname>Mursi</surname></string-name> and <string-name><given-names>A. M.</given-names> <surname>Almalki</surname></string-name></person-group>, &#x201C;<article-title>ArCyb: A robust machine-learning model for Arabic cyberbullying tweets in Saudi Arabia</article-title>,&#x201D; <source>Int. J. Adv. Comput. Sci. Appl.</source>, vol. <volume>14</volume>, no. <issue>9</issue>, pp. <fpage>1059</fpage>&#x2013;<lpage>1067</lpage>, <year>2023</year>. doi: <pub-id pub-id-type="doi">10.14569/IJACSA.2023.01409110</pub-id>.</mixed-citation></ref>
<ref id="ref-54"><label>[54]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><collab>Atta-ur-Rahman</collab>, <string-name><given-names>S.</given-names> <surname>Dash</surname></string-name>, <string-name><given-names>A. K.</given-names> <surname>Luhach</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Chilamkurti</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Baek</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Nam</surname></string-name></person-group>, &#x201C;<article-title>A neuro-fuzzy approach for user behaviour classification and prediction</article-title>,&#x201D; <source>J. Cloud Comput.</source>, vol. <volume>8</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>15</lpage>, <year>2019</year>. doi: <pub-id pub-id-type="doi">10.1186/s13677-019-0144-9</pub-id>.</mixed-citation></ref>
<ref id="ref-55"><label>[55]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>V.</given-names> <surname>Balakrisnan</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Kaity</surname></string-name></person-group>, &#x201C;<article-title>Cyberbullying detection and machine learning: A systematic literature review</article-title>,&#x201D; <source>Artif. Intell. Rev.</source>, vol. <volume>56</volume>, no. <issue>Suppl 1</issue>, pp. <fpage>1375</fpage>&#x2013;<lpage>1416</lpage>, <year>2023</year>. doi: <pub-id pub-id-type="doi">10.1007/s10462-023-10553-w</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>