<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">IASC</journal-id>
<journal-id journal-id-type="nlm-ta">IASC</journal-id>
<journal-id journal-id-type="publisher-id">IASC</journal-id>
<journal-title-group>
<journal-title>Intelligent Automation &#x0026; Soft Computing</journal-title>
</journal-title-group>
<issn pub-type="epub">2326-005X</issn>
<issn pub-type="ppub">1079-8587</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">39687</article-id>
<article-id pub-id-type="doi">10.32604/iasc.2023.039687</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Ensemble-Based Approach for Efficient Intrusion Detection in Network Traffic</article-title>
<alt-title alt-title-type="left-running-head">Ensemble-Based Approach for Efficient Intrusion Detection in Network Traffic</alt-title>
<alt-title alt-title-type="right-running-head">Ensemble-Based Approach for Efficient Intrusion Detection in Network Traffic</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Almomani</surname><given-names>Ammar</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref><email>ammarnav6@bau.edu.jo</email></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Akour</surname><given-names>Iman</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Manasrah</surname><given-names>Ahmed M.</given-names></name><xref ref-type="aff" rid="aff-4">4</xref><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Almomani</surname><given-names>Omar</given-names></name><xref ref-type="aff" rid="aff-6">6</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Alauthman</surname><given-names>Mohammad</given-names></name><xref ref-type="aff" rid="aff-7">7</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Abdullah</surname><given-names>Esra&#x2019;a</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Shwait</surname><given-names>Amaal Al</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>Sharaa</surname><given-names>Razan Al</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<aff id="aff-1"><label>1</label><institution>School of Computing, Skyline University College, University City of Sharjah, P. O. Box 1797</institution>, <addr-line>Sharjah</addr-line>, <country>United Arab Emirates</country></aff>
<aff id="aff-2"><label>2</label><institution>IT-Department-Al-Huson University College, Al-Balqa Applied University, P. O. Box 50</institution>, <addr-line>Irbid</addr-line>, <country>Jordan</country></aff>
<aff id="aff-3"><label>3</label><institution>Information Systems Department, College of Computing &#x0026; Informatics, University of Sharjah</institution>, <country>United Arab Emirates</country></aff>
<aff id="aff-4"><label>4</label><institution>Comp. Info Sciences (CIS) Division, Higher Colleges of Technology</institution>, <addr-line>Sharjah</addr-line>, <country>United Arab Emirates</country></aff>
<aff id="aff-5"><label>5</label><institution>Computer Sciences Department, Yarmouk University</institution>, <addr-line>Irbid</addr-line>, <country>Jordan</country></aff>
<aff id="aff-6"><label>6</label><institution>Computer Network and Information Systems Department, The World Islamic Sciences and Education University</institution>, <addr-line>Amman, 11947</addr-line>, <country>Jordan</country></aff>
<aff id="aff-7"><label>7</label><institution>Department of Information Security, Faculty of Information Technology, University of Petra</institution>, <addr-line>Amman</addr-line>, <country>Jordan</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Ammar Almomani. Email: <email>ammarnav6@bau.edu.jo</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2023</year></pub-date>
<pub-date date-type="pub" publication-format="electronic"><day>23</day>
<month>6</month>
<year>2023</year></pub-date>
<volume>37</volume>
<issue>2</issue>
<fpage>2499</fpage>
<lpage>2517</lpage>
<history>
<date date-type="received">
<day>11</day>
<month>2</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>5</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2023 Almomani et al.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Almomani et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_IASC_39687.pdf"></self-uri>
<abstract>
<p>The exponential growth of Internet and network usage has necessitated heightened security measures to protect against data and network breaches. Intrusions, executed through network packets, pose a significant challenge for firewalls to detect and prevent due to the similarity between legitimate and intrusion traffic. The vast network traffic volume also complicates most network monitoring systems and algorithms. Several intrusion detection methods have been proposed, with machine learning techniques regarded as promising for dealing with these incidents. This study presents an Intrusion Detection System Based on Stacking Ensemble Learning base (Random Forest, Decision Tree, and k-Nearest-Neighbors). The proposed system employs pre-processing techniques to enhance classification efficiency and integrates seven machine learning algorithms. The stacking ensemble technique increases performance by incorporating three base models (Random Forest, Decision Tree, and k-Nearest-Neighbors) and a meta-model represented by the Logistic Regression algorithm. Evaluated using the UNSW-NB15 dataset, the proposed IDS gained an accuracy of 96.16% in the training phase and 97.95% in the testing phase, with precision of 97.78%, and 98.40% for taring and testing, respectively. The obtained results demonstrate improvements in other measurement criteria.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Intrusion detection system (IDS)</kwd>
<kwd>machine learning techniques</kwd>
<kwd>stacking ensemble</kwd>
<kwd>random forest</kwd>
<kwd>decision tree</kwd>
<kwd>k-nearest-neighbor</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Network security has become vital to its successful operation with the emerging Internet and communication technologies. Therefore, organizations are forced to invest in the security of their sensitive information and functions by adopting different security controls such as firewalls, anti-virus software, and Intrusion Detection Systems (IDS) to ensure the security of the network and all associated assets. The IDS is essential for identifying abnormal traffic and notifying the network administrator [<xref ref-type="bibr" rid="ref-1">1</xref>].</p>
<p>Machine learning techniques have garnered substantial popularity in network security over the last decade due to their ability to learn relevant features from network data and perform accurate categorization based on learned patterns [<xref ref-type="bibr" rid="ref-2">2</xref>]. Furthermore, due to its deep architecture, Deep Learning (DL)-based IDS rely on the automated learning of intricate characteristics from raw data [<xref ref-type="bibr" rid="ref-3">3</xref>].</p>
<p>Ensemble learning, on the other hand, is a machine learning paradigm in which several models, such as classifiers or experts, are developed and integrated strategies to address a specific computational intelligence issue. Ensemble learning is primarily used to improve a model&#x0027;s performance (Classification, prediction, function approximation, etc.) or to lessen the risk of an unintentional poor model selection. The architecture of the ensemble model is depicted in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>The basic architecture of ensemble classifiers</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="IASC_39687-fig-1.tif"/>
</fig>
<p>This paper proposes an IDS that utilizes a diversity of individual experts to analyze network traffic requests and responses. The system tracks the activity from when it enters the network to when it causes harm or performs unauthorized actions. The system achieves classifier diversity by employing different training parameters for each classifier, allowing individual classifiers to produce various decision boundaries. Combining these diverse errors leads to a lower overall error and higher accuracy than other A.I. and DL algorithms.</p>
<p>The research aims to develop an effective and efficient IDS (IDS) for detecting security breaches in complex network traffic. The objectives of this research are:
<list list-type="bullet">
<list-item>
<p>To evaluate the performance of several machine learning algorithms for intrusion detection.</p></list-item>
<list-item>
<p>To use stacking ensemble learning to improve the performance of the IDS.</p></list-item>
<list-item>
<p>To evaluate the proposed IDS system using the UNSW-NB15 dataset and to compare its performance with other existing IDS.</p></list-item>
</list></p>
<p>The contributions of this research are:
<list list-type="bullet">
<list-item>
<p>The proposed IDS system uses the stacking ensemble learning technique to improve the performance of intrusion detection.</p></list-item>
<list-item>
<p>The system is evaluated using the UNSW-NB15 dataset and shows high accuracy and improved performance compared to other IDS.</p></list-item>
<list-item>
<p>The study provides valuable insights into using machine learning algorithms and stacking ensemble learning for intrusion detection, which can contribute to advancing this field.</p></list-item>
</list></p>
<p>This paper is organized as follows: Section 2 presents the related work. Section 3 proposed an intrusion detection model to which different machine learning techniques are applied is described. While Section 4 provides the Implementation, and the results are discussed. Finally, Section 5 concludes the paper and presents future work.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Work</title>
<p>Jim Anderson first proposed the idea of IDS in 1980 [<xref ref-type="bibr" rid="ref-4">4</xref>]. Since then, many IDS products have been developed and matured to meet network security needs [<xref ref-type="bibr" rid="ref-5">5</xref>]. A combination of &#x201C;intrusion&#x201D; and &#x201C;detection systems&#x201D; is known as an IDS. The term &#x201C;intrusion&#x201D; refers to unlawful access to computer systems or a network&#x2019;s internal data to compromise its integrity, confidentiality, or availability [<xref ref-type="bibr" rid="ref-6">6</xref>]. In contrast, the detection system serves as a safeguard against illicit activities. Because of this, IDS is a security tool that constantly monitors host and network traffic to detect any suspicious behavior that violates the security policy and compromises the network&#x2019;s confidentiality, integrity, and availability. It is connected to a network adapter configured with port mirroring technology, as shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>.</p>
<p>The IDS will flag the host or network administrator&#x2019;s malicious behavior. Scientists have investigated machine learning and DL approaches to meet the needs of successful IDS. Aiming to acquire meaningful information from large datasets, both Machine Learning (ML) and DL) are being explored. Over the past decade, the development of extremely powerful Graphics Processing Units (GPUs) has led to the widespread adoption of these technologies in network security [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-8">8</xref>]. ML and DL are powerful methods for network traffic analysis and traffic prediction. To derive valuable information from network traffic, ML-based IDS significantly relies on engineering features [<xref ref-type="bibr" rid="ref-2">2</xref>]. In contrast, DL-based IDS relies on automatically learning complex features from raw data due to its deep architecture [<xref ref-type="bibr" rid="ref-3">3</xref>].</p>
<p>When designing an IDS, four functional modules,1-Event-boxes; 2-Database-boxes; 3-Analysis-boxes; and 4-Routine boxes, are used to build the overall architecture, as shown in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>. For the most part, Event-Boxes are sensors that monitor the system and gather data for subsequent analysis. This gathered data must be kept for processing. Data-base-box elements serve this purpose by storing the information received from the event-Boxes. The Analysis-boxes processing module is where harmful conduct is detected by examining events. The most critical step is to stop the hostile conduct once it has been identified. On the other hand, the Response-boxes take action immediately if any intrusion is detected. In <xref ref-type="fig" rid="fig-2">Fig. 2</xref>, an example of the IDS framework is shown [<xref ref-type="bibr" rid="ref-1">1</xref>]. A Host-Based IDS (HIDS) and a Network-based IDS (NIDS) can be characterized by information source Event-Boxes (NIDS). System calls and process identifiers are the focus of HIDS, while network events are the focus of NIDS (I.P. address, protocols, service ports, traffic volume, etc.). IDS can be divided into signature-based IDS (misuse-based) and anomaly-based IDS based on the analysis done in Analysis-boxes. (Signature-based IDS) SIDS utilizes a database of known attack signatures to detect intrusions by comparing captured data against the database [<xref ref-type="bibr" rid="ref-9">9</xref>]. Only known assaults can be detected using this method; new threats cannot be detected using this method (previously unseen attacks). Signature-based IDS has a significantly reduced false-positive rate.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>IDS(IDS)&#x2014;monitors network traffic [<xref ref-type="bibr" rid="ref-13">13</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="IASC_39687-fig-2.tif"/>
</fig>
<p>On the other hand, Anomaly-based IDS (AIDS) aims to understand the system&#x0027;s usual behavior and establishes a threshold. An anomaly alarm is sounded when a certain observation deviates from its regular pattern [<xref ref-type="bibr" rid="ref-10">10</xref>]. Anomaly-based IDS is useful for detecting previously unnoticed assaults since it seeks to find unusual occurrences. In contrast to SIDS, AIDS has a greater false-positive rate for intrusion detection [<xref ref-type="bibr" rid="ref-1">1</xref>].</p>
<p>To effectively detect network-based attacks, many researchers attempted to combine Stack ensemble learning, a collection of ML algorithms. For example, the stacked ensemble technique presented was assessed using.</p>
<p>The performance of the suggested method was compared to that of other well-known ML algorithms such as Artificial neural network ANN, CART, random forest, and Support vector machines SVM. The experimental results show that stacked ensemble learning is a proper technique for classifying network-based attacks. Similarly, the authors in [<xref ref-type="bibr" rid="ref-8">8</xref>] applied a group of learning algorithms over the UNSW-NB15 dataset using the stacking classifier method. The mixed method for feature selection also includes Lasso regression using SVM. The accuracy of the Lasso was evaluated using the R2 score evaluation, which was 59%. Similar to the research in [<xref ref-type="bibr" rid="ref-11">11</xref>], Gao et al. [<xref ref-type="bibr" rid="ref-12">12</xref>] analyze the NSL-KDD dataset by exploring different training ratios and creating multiple decision trees to conclude their MultiTree detection algorithm with adaptive voting in multiple classifier algorithms. <xref ref-type="fig" rid="fig-2">Fig. 2</xref> shows the IDS for network traffic monitoring, and <xref ref-type="fig" rid="fig-3">Fig. 3</xref> show IDS Architecture.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Common intrusion detection architecture for IDS [<xref ref-type="bibr" rid="ref-1">1</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="IASC_39687-fig-3.tif"/>
</fig>
<p>For this reason, the authors adopted different classifiers, including Decision trees, random forests, K-Nearest Neighbors KNN, DNN. The adaptive voting algorithm increases the detection accuracy from 84.1% to 85.2%. Therefore, Larriva-Novo et al. [<xref ref-type="bibr" rid="ref-14">14</xref>] suggested using a Dynamic classifier multiclass to obtain the best capabilities for each cyber-attack detection model. They tested their approach on a UNSW-NB15 dataset, in which the dataset was split into training data (75%) and tests (25%). The dynamic classifier improved results by 5.3% and 2.3% compared to the best static model for the balanced and unbalanced datasets. Their work is achieving a noticeable increase in performance in terms of detection rate for IDS based on multiple classes of attacks.</p>
<p>Slay [<xref ref-type="bibr" rid="ref-15">15</xref>] proposed using a Hyper-Clique-based Improved Binary Gravitational Algorithm (HC-IBGSA) to identify the optimal model parameters and feature set for SVM. The proposed approach was evaluated using two benchmark intrusion datasets, the NSL-KDD CUP and the UNSW-NB15 dataset. The evaluation was carried out over two feature sets for SVM training: (1) SVM trained with all features and (2) SVM trained with the optimal features obtained from HC-IBGSA. The proposed approach shows a 94.11% classification accuracy using the UNSW NB15 dataset with features extracted using the HC-IBGSA algorithms. Similarly, Slay [<xref ref-type="bibr" rid="ref-15">15</xref>] presents a feature selection for rare cyber-attacks based on the UNSW-NB15 dataset using the Random Committee technique. The proposed approach evaluation of the multiclass Classification obtains an accuracy of 99.94%. However, the high accuracy rate was reported as the best case for a work attack.</p>
<p>Consequently, Correctly classifying network flows as benign or malicious traffic is the way the authors [<xref ref-type="bibr" rid="ref-16">16</xref>] have adopted. Their approach depends on classifying network traffic flows using R.F., MLP, and LSTM. The proposed approach evaluations using the CIDDS-001 dataset yield 99.94% accuracy. As a result, many researchers have started to adopt various classifiers into their intrusion detection techniques to accommodate the different types of input data. For instance, In [<xref ref-type="bibr" rid="ref-17">17</xref>] the authors attempted to unite the strengths of SIDS and an AIDS-based IDS into a new hybrid IDS system (HIDS). The new HIDS combines the C5 decision tree classifier and a single class support vector machine (OC-SVM). Using the Network Security Lab Knowledge Discovery in databases (NSL-KDD) and Australian Defense Force Academy (ADFA) datasets, the authors confirm that they achieve low alarm rates.</p>
<p>To summarize, proposing an intrusion detection model would entail using multiple datasets to demonstrate detection capability and extensibility in various environmental settings. <xref ref-type="table" rid="table-1">Table 1</xref> shows that the NSL-KDD, KDD Cup 99, and UNSW-NB15 datasets are widely used in IDS research.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Summary of selected studies utilizing ensemble methods for IDSs</title>
</caption>
 
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Reference</th>
<th>Algorithm</th>
<th>Accuracy</th>
<th>Classifier</th>
<th>Dataset</th>
<th>Adv.</th>
<th>Limitation</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td>Jing<break/>et al. [<xref ref-type="bibr" rid="ref-18">18</xref>]</td>
<td>SVM</td>
<td>85.99%<break/>75.77%</td>
<td>Binary<break/>Multiclass</td>
<td>UNSW-NB15</td>
<td>The proposed SVM method accuracy outperforms the Naive Bayes (N.B.) method with a score of 75.77%.</td>
<td>Compared to the UNSW-NB15 dataset, the KDDCUP99 dataset lacks some common examples for NIDS evaluation.</td>
</tr>
<tr>
<td>Ahmad<break/> et al. [<xref ref-type="bibr" rid="ref-19">19</xref>]</td>
<td>SVM<break/>RF<break/>ELM</td>
<td>99%<break/>97.5%<break/>99.5%</td>
<td>Binary</td>
<td>NSL&#x2013;KDD</td>
<td>These techniques are well-known for their classifiers, and ELM outperforms other approaches (SVM, R.F.).</td>
<td>SVM outperforms other approaches on small datasets, whereas EML outperforms others on large datasets.</td>
</tr>
<tr>
<td>Rajadurai<break/>et al. [<xref ref-type="bibr" rid="ref-11">11</xref>]</td>
<td>Ensemble<break/>ANN<break/>RF<break/>Na&#x00EF;ve Bayes<break/>SVM</td>
<td>91.06%<break/>97.74%<break/>74.00%<break/>74.40%<break/>74.00%</td>
<td>Binary</td>
<td>NSL-KDD</td>
<td>Staked ensemble learning is suitable for classifying attacks, and the proposed system outperforms other intrusion detection models in terms of accuracy.</td>
<td>The precision and recall values of the RNN and ANN approaches are significantly lower than those of the proposed approach.</td>
</tr>
<tr>
<td>Abirami <break/>et al. [<xref ref-type="bibr" rid="ref-8">8</xref>]</td>
<td>RF<break/>SVM<break/>Naive Bayes<break/>Logistic<break/>regression</td>
<td>93%<break/>85%<break/>79%<break/>80%</td>
<td>Binary</td>
<td>KDD Cup99<break/>NSK-KDD<break/>Kyoto2006</td>
<td>The clustering classifier and ensemble algorithm yielded better results.</td>
<td>The decision tree classification algorithm and the regression algorithms achieved less precision.</td>
</tr>
<tr>
<td>Gao<break/>et al. [<xref ref-type="bibr" rid="ref-12">12</xref>]</td>
<td>DeciTree<break/>RF<break/>LR<break/>KNN<break/>DNN<break/>Adaboost<break/>SVM</td>
<td>73.58%<break/>79.71%<break/>74.09%<break/>76.02%<break/>81.6%</td>
<td>Binary</td>
<td>NSL-KDD</td>
<td>The ensemble model significantly improves detection accuracy.</td>
<td>The deep neural network has some advantages in detection, but it takes time.</td>
</tr>
<tr>
<td>Ring<break/>et al. [<xref ref-type="bibr" rid="ref-20">20</xref>]</td>
<td>NBTree<break/>Fuzzy<break/>SVM<break/>FS&#x002B;GAR-<break/>forest<break/>TDTC<break/>FSSL<break/>EM-FS<break/>FSSL-EL<break/>TSE-IDS<break/>NBTree<break/>FSSL<break/>FSSL-EL<break/>TSE-IDS</td>
<td>82.02%<break/>82.74%<break/>82.37%<break/>85.05%<break/>84.86%<break/>84.25%<break/>84.54%<break/>85.79%<break/>66.16%<break/>68.82%<break/>71.29%<break/>72.52%<break/>87.37%<break/>73.57%</td>
<td>Binary</td>
<td>NSL-KDD<break/>UNSW-NB15</td>
<td>Under several criteria, the proposed CFS-BA-Ensemble method outperforms other relevant techniques.</td>
<td>Difficult to identify high attack detection rates (ADR) while minimizing false alarm rates (FAR).</td>
</tr>
<tr>
<td>Bamakan<break/>et al. [<xref ref-type="bibr" rid="ref-21">21</xref>]</td>
<td>XGBoost<break/>KNN<break/>Logistic Regression Stacking</td>
<td>83.54%<break/>84.00%<break/>63.54%<break/>85.42%</td>
<td>Binary<break/>Multiclass</td>
<td>MLPAT<break/>ELM</td>
<td>APT attacks are conducted with high planning levels and a high degree of target recognition.</td>
<td>The cost of implementing APT is too high.</td>
</tr>
<tr>
<td>Larriva-Novo<break/>et al. [<xref ref-type="bibr" rid="ref-14">14</xref>]</td>
<td>Dynamic<break/>Classifier</td>
<td>87.6%</td>
<td>Binary<break/>Multiclass</td>
<td>UNSW-NB15</td>
<td>The dynamic classifier model improves the detection accuracy<break/> of each model.</td>
<td>TPR can be reduced by up to 40% using category exploit.</td>
</tr>
<tr>
<td>Pang<break/>et al. [<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
<td>CFS-BA</td>
<td>99.81%</td>
<td>Binary<break/>Multiclass</td>
<td>NSL-KDD<break/>AWID<break/>ClC-<break/>IDS2017</td>
<td>The proposed CFS-BA-Ensemble method outperforms other approaches on different metrics.</td>
<td>Efficient while keeping FAR under control</td>
</tr>
<tr>
<td>Sindhu <break/>et al. [<xref ref-type="bibr" rid="ref-23">23</xref>]</td>
<td>XGBoost<break/>KNN<break/>Logistic Regression<break/>Stacking</td>
<td>83%<break/>84%<break/>52%<break/>85%</td>
<td>Binary</td>
<td>KDD&#x2019;99</td>
<td>The stack classifier achieved the best result compared to otherclassifiers.</td>
<td>Logisticregression is the worst inaccuracy.</td>
</tr>
<tr>
<td>Lee<break/>et al. [<xref ref-type="bibr" rid="ref-24">24</xref>]</td>
<td>DNN<break/>SHAP<break/>BRCG<break/>CEM</td>
<td>90.82%<break/>87.96%<break/>82.71%<break/>92.82%</td>
<td>Binary</td>
<td>NSL-KDD<break/>ClC-<break/>IDS2017<break/>KDDT</td>
<td>It gives a much better insight to the security analyst on why the alert was flagged.</td>
<td>The model may learn that demand leads to poverty performance.</td>
</tr>
<tr>
<td>Raman<break/>et al. [<xref ref-type="bibr" rid="ref-25">25</xref>]</td>
<td>C4.5<break/>Na&#x00EF;ve Bayes<break/>RF<break/>Multilayer Perception<break/>SVM<break/>CART<break/>KNN</td>
<td>81%<break/>76.56%<break/>80.67%<break/>77.41%<break/>69.52%<break/>80.3%<break/>79.4%</td>
<td>Binary<break/>Multiclass</td>
<td>NSL-KDD CIDDS-<break/>001</td>
<td>Possibility of gaining access to a high level of electronic resilience against malicious activity and unauthorized identification.</td>
<td>These technologies may be incapable of generating and updating a new malware signal due to high alarms or low detection rates.</td>
</tr>
<tr>
<td>Raman<break/>et al. [<xref ref-type="bibr" rid="ref-26">26</xref>]</td>
<td>HC-IBGSA</td>
<td>94.11%</td>
<td>Binary<break/>Multiclass</td>
<td>NSL-KDD,<break/>UNSW-NB15</td>
<td>Using more recent IDS datasets to assess algorithm performance before and after feature selection</td>
<td>Not using different classifiers, only SVM</td>
</tr>
<tr>
<td>Slay [<xref ref-type="bibr" rid="ref-15">15</xref>]</td>
<td>Feedforward<break/>NN</td>
<td>99.94%</td>
<td>Binary<break/>Multiclass</td>
<td>UNSW-NB15</td>
<td>The DL model achieves very high accuracy.</td>
<td>Traditional MLalgorithms are inefficient at classifying Network Intrusions.</td>
</tr>
<tr>
<td>Khraisat <break/>et al. [<xref ref-type="bibr" rid="ref-16">16</xref>]</td>
<td>RF<break/>MLP<break/>LSTM</td>
<td>99.94%</td>
<td>Attack type</td>
<td>CIDDS-001</td>
<td>The multi-flow method is appropriate for detecting anomalies in the CIDDS-001 dataset.</td>
<td>As the length of the sequence increases, the radiofrequency decreases dramatically.</td>
</tr>
<tr>
<td>Rashid<break/>et al. [<xref ref-type="bibr" rid="ref-27">27</xref>]</td>
<td>k-NN<break/>Na&#x00EF;ve Bayes<break/>SVM<break/>NN<break/>DNN<break/>Auto-encoders</td>
<td>99.80%<break/>98.60%<break/>100%<break/>99.90%<break/>99.90%<break/>98.60%</td>
<td>Binary</td>
<td>NSL-KDD<break/>CIDDS-001</td>
<td>SVM, DNN, and k-NN classifiers all perform similarly.</td>
<td>Since the multiclass Classification was not addressed, the types of attacks in the CIDDS-001 dataset cannot be discovered.</td>
</tr>
<tr>
<td>Rababah <break/>et al. [<xref ref-type="bibr" rid="ref-17">17</xref>]</td>
<td>C4.5<break/>Na&#x00EF;ve Bayes<break/>SVM<break/>CART<break/>KNN</td>
<td>81%<break/>76.56%<break/>80.67%<break/>69.52%<break/>80.3%</td>
<td>Binary</td>
<td>NSL-KDD</td>
<td>Compared to SIDS and AIDS, HIDS has a higher detection rate and lower alarm rate.</td>
<td>Single algorithms give in accurate results.</td>
</tr>
<tr>
<td>Yang<break/>et al. [<xref ref-type="bibr" rid="ref-28">28</xref>]</td>
<td>RBFN<break/>Na&#x00EF;ve Bayes<break/>DT<break/>RI<break/>K-NN</td>
<td>92.17%<break/>91.23%<break/>91.38%<break/>91.81%<break/>91.24%</td>
<td>Binary<break/>Multiclass</td>
<td>CIDDS-<break/>001</td>
<td>Correlation rules and group analysis are used to detect illegal activities in database usage patterns.</td>
<td>The emphasis was on reducing false alarms rather than increasing the detection rate.</td>
</tr>
<tr>
<td>Gautam <break/>et al. [<xref ref-type="bibr" rid="ref-29">29</xref>]</td>
<td>KNN<break/>SVM<break/>DT<break/>RF<break/>ET<break/>XGBoost<break/>Stacking<break/>FSXGBoost<break/>FS Stacking</td>
<td>96.6%<break/>98.01%<break/>99.72%<break/>98.37%<break/>93.43%<break/>99.78%<break/>99.86%<break/>99.7%<break/>99.82%</td>
<td>Binary</td>
<td>NSL-KDD<break/>CIDDS-001</td>
<td>The proposed system has a high detection rate and a low computing cost.</td>
<td>Require high computation and storage requirements.</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3">
<label>3</label>
<title>Data Acquisition (Collection, Information Gathering)</title>
<p>To test and evaluate the proposed approach, we have used the UNSW-NB15 dataset from ACCS [<xref ref-type="bibr" rid="ref-15">15</xref>], a modern NIDS benchmark data set for Network IDS. The dataset has 2.5 million records that are divided into 45 features. We have modified the original UNSW-NB15 dataset to reduce the number of features to 43 instead of 45, including flow-based and packet-based features. These features are further subdivided into four categories: content, fundamental, flow, and time-based features. The number of selected data instances from the UNSW-NB15 dataset is 257,673, divided into training data instances (175,341 records) and testing data instances (82,332 records). The record distribution and class distribution of the UNSW-NB15 dataset are shown in <xref ref-type="fig" rid="fig-4">Fig. 4</xref> [<xref ref-type="bibr" rid="ref-30">30</xref>].</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>UNSW-NB15 dataset distribution</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="IASC_39687-fig-4.tif"/>
</fig>
<p>The UNSW-NB15 dataset provides two labeled features (i.e., cat-attack and label). Label characteristics were only utilized when the data was either normal or attacked (binary data). The features list and their names are given in <xref ref-type="table" rid="table-2">Table 2</xref>.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Features of the UNSW-NB15 [<xref ref-type="bibr" rid="ref-15">15</xref>]</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>No.</th>
<th>Name</th>
<th>T</th>
<th>group</th>
<th>Description</th>
</tr>
</thead><tbody valign="top">
<tr>
<td>1</td>
<td>dur</td>
<td>F</td>
<td>Basic</td>
<td>Total record length</td>
</tr>
<tr>
<td>2</td>
<td>Proto</td>
<td>N</td>
<td>Flow</td>
<td>Transaction protocol</td>
</tr>
<tr>
<td>3</td>
<td>service</td>
<td>N</td>
<td>Basic</td>
<td>HTTP, FTP, SSH, DNS .., else (-)</td>
</tr>
<tr>
<td>4</td>
<td>state</td>
<td>N</td>
<td>Basic</td>
<td>The state and its dependent protocol, e.g., ACC, CLO, else (-)</td>
</tr>
<tr>
<td>5</td>
<td>spkts</td>
<td>I</td>
<td>Basic</td>
<td>Count of packets sent from the source to the destination</td>
</tr>
<tr>
<td>6</td>
<td>dpkts</td>
<td>I</td>
<td>Basic</td>
<td>Count of packets from source to destination</td>
</tr>
<tr>
<td>7</td>
<td>sbytes</td>
<td>I</td>
<td>Basic</td>
<td>Bytes from the source to the destination</td>
</tr>
<tr>
<td>8</td>
<td>dbytes</td>
<td>I</td>
<td>Basic</td>
<td>Source bytes to destination bytes</td>
</tr>
<tr>
<td>9</td>
<td>Rate</td>
<td></td>
<td>Basic</td>
<td>This term is in the training/test sets but not defined anywhere.</td>
</tr>
<tr>
<td>10</td>
<td>sttl</td>
<td>I</td>
<td>Basic</td>
<td>Source to live time of destination.</td>
</tr>
<tr>
<td>11</td>
<td>dttl</td>
<td>I</td>
<td>Basic</td>
<td>Time to live in the destination.</td>
</tr>
<tr>
<td>12</td>
<td>sload</td>
<td>F</td>
<td>Basic</td>
<td>Bits of source per second.</td>
</tr>
<tr>
<td>13</td>
<td>dload</td>
<td>F</td>
<td>Basic</td>
<td>Bits per second destination.</td>
</tr>
<tr>
<td>14</td>
<td>sloss</td>
<td>I</td>
<td>Basic</td>
<td>Retransmitted or dropped source packets.</td>
</tr>
<tr>
<td>15</td>
<td>dloss</td>
<td>I</td>
<td>Basic</td>
<td>Retransmitted or dropped destination packets.</td>
</tr>
<tr>
<td>16</td>
<td>sintpkt</td>
<td>F</td>
<td>Time</td>
<td>Time of arrival for the interpacket source (mSec).</td>
</tr>
<tr>
<td>17</td>
<td>dintpkt</td>
<td>F</td>
<td>Time</td>
<td>Time of arrival between packets (mSec).</td>
</tr>
<tr>
<td>18</td>
<td>sjit</td>
<td>F</td>
<td>Time</td>
<td>Jitter Source (mSec).</td>
</tr>
<tr>
<td>19</td>
<td>djit</td>
<td>F</td>
<td>Time</td>
<td>Jitter Destination (mSec).</td>
</tr>
<tr>
<td>20</td>
<td>swin</td>
<td>I</td>
<td>Content</td>
<td>Advertising Content Source TCP window.</td>
</tr>
<tr>
<td>21</td>
<td>dwin</td>
<td>I</td>
<td>Content</td>
<td>Advertisement in the TCP window for the destination</td>
</tr>
<tr>
<td>22</td>
<td>stcpb</td>
<td>I</td>
<td>Content</td>
<td>TCP sequence number from the source</td>
</tr>
<tr>
<td>23</td>
<td>dtcpb</td>
<td>I</td>
<td>Content</td>
<td>TCP sequence number for the destination</td>
</tr>
<tr>
<td>24</td>
<td>tcprtt</td>
<td>F</td>
<td>Time</td>
<td>The TCP&#x2019;s &#x2018;synack&#x2019; and &#x2018;ackdat&#x2019; are added together.</td>
</tr>
<tr>
<td>25</td>
<td>synack</td>
<td>F</td>
<td>Time</td>
<td>The interval between the TCP packets SYN and SYN ACK.</td>
</tr>
<tr>
<td>26</td>
<td>ackdat</td>
<td>F</td>
<td>Time</td>
<td>The interval between the TCP&#x2019;s SYN ACK and ACK packets.</td>
</tr>
<tr>
<td>27</td>
<td>smeansz</td>
<td>I</td>
<td>Content</td>
<td>Means the flow packet&#x2019;s src size.</td>
</tr>
<tr>
<td>28</td>
<td>dmeansz</td>
<td>I</td>
<td>Content</td>
<td>The average size of the dst sent flow packets.</td>
</tr>
<tr>
<td>29</td>
<td>trans_depth</td>
<td>I</td>
<td>Content</td>
<td>The connection&#x2019;s depth in the http request/response transaction.</td>
</tr>
<tr>
<td>30</td>
<td>res_bdy_len</td>
<td>I</td>
<td>Content</td>
<td>The magnitude of the data transmitted by the server&#x2019;s http service.</td>
</tr>
<tr>
<td>31</td>
<td>ct_srv_src</td>
<td>I</td>
<td>Connection</td>
<td>No, this is the 100th time in 100 connections with the same service and source address.</td>
</tr>
<tr>
<td>32</td>
<td>ct_state_ttl</td>
<td>I</td>
<td>General</td>
<td>No, for each state, based on a defined source/destination value range. to live.</td>
</tr>
<tr>
<td>33</td>
<td>ct_dst_ltm</td>
<td>I</td>
<td>Connection</td>
<td>No. connections of the same destination address Last time in 100 connections.</td>
</tr>
<tr>
<td>34</td>
<td>ct_src_dport_ltm</td>
<td>I</td>
<td>Connection</td>
<td>No Last time there were no connections with the same source address and destination port at 100 connections.</td>
</tr>
<tr>
<td>35</td>
<td>ct_dst_sport_ltm</td>
<td>I</td>
<td>Connection</td>
<td>No of the Last time in 100 connections no connections of the same destination address and source port.</td>
</tr>
<tr>
<td>36</td>
<td>ct_dst_src_ltm</td>
<td>I</td>
<td>Connection</td>
<td>No of the Last time there were 100 connections with the same source and destination address.</td>
</tr>
<tr>
<td>37</td>
<td>is_ftp_login</td>
<td>B</td>
<td>General</td>
<td>If a user and password are used for the ftp session, then another 1 is 0.</td>
</tr>
<tr>
<td>38</td>
<td>ct_ftp_cmd</td>
<td>I</td>
<td>General</td>
<td>There are no flows with ftp command.</td>
</tr>
<tr>
<td>39</td>
<td>ct_flw_http_mthd</td>
<td>I</td>
<td>General</td>
<td>No. flows with methods like Get and Post in HTTP.</td>
</tr>
<tr>
<td>40</td>
<td>ct_src_ltm</td>
<td>I</td>
<td>Connection</td>
<td>The number of connections with the same source address in the last time 100 connections.</td>
</tr>
<tr>
<td>41</td>
<td>ct_srv_dst</td>
<td>I</td>
<td>Connection</td>
<td>No. of the Last time connections of the same source address in 100 connections.</td>
</tr>
<tr>
<td>42</td>
<td>is_sm_ips_ports</td>
<td>B</td>
<td>General</td>
<td>This variable will have a value of 1 other 0 if the source is identical to destination I.P. and port numbers.</td>
</tr>
<tr>
<td>43</td>
<td>attack_cat</td>
<td>N</td>
<td></td>
<td>Every type of assault is given a name. This data collection has nine categories (e.g., Fussers, Analysis, Backdoors, DoS, Exploits, Generic, Reconnaissance, Shellcode, and Worms).</td>
</tr>
<tr>
<td>44</td>
<td>Label</td>
<td>B</td>
<td></td>
<td>Normal records have a value of 0 while attack records have a value of 1.</td>
</tr>
</tbody>
</table>
<table-wrap-foot><fn id="table-2fn1">
<p>Note: Type (T.), N: nominal, I: integer, F: float, T: timestamp, and B: binary.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The table above shows the different types of available features. These are called nominal features and are grouped into proto, service, state, and attack_cat features. The proto feature has 133 values related to the TCP and UDP protocols. The service feature has 13 values related to the different network services, such as DNS, HTTP, SMTP, FTP data, FTP, SSH, POP3, DHCP, SNMP, SSL, IRC, and radius, as illustrated in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>. The state feature has 9 values related to different transport layer protocol flags such as INT, FIN, CON, REQ, RST, ECO, PAR, no, and URN, as illustrated in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>. The other category has a &#x201C;-&#x201D; followed by a long DNS name. For the conversion of nominal features into a numerical format, we employed one-hot encoding. This method represents each category as a binary vector, which is compatible with our feature selection process and the machine learning algorithms used in this study.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>The distribution of the service feature</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="IASC_39687-fig-5.tif"/>
</fig><fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>The distribution of the state feature</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="IASC_39687-fig-6.tif"/>
</fig>
</sec>
<sec id="s4">
<label>4</label>
<title>Proposed Ensemble Learning System Description</title>
<p>Ensemble learning is a widely used machine learning technique that combines the capabilities of individual classifiers to create a classification model with superior overall Classification or predictive power. In the context of intrusion detection systems (IDS), ensemble learning has demonstrated better performance compared to standalone classifiers. This study aims to develop a robust network intrusion detection system using an ensemble classifier approach. The proposed solution requires pre-processing of datasets for training purposes. In this work, we primarily focus on the UNSW-NB15 dataset, which comprises 45 features, including four nominal and 41 numerical features, after converting them to numerical values. Data normalization is essential, as the dimensions and units used in data collection vary, necessitating the scaling of different feature values within a specific range. In this study, we employ the maximum and minimum algorithms to normalize the data. The MinMaxScaler algorithm scales and translates each feature individually to lie between a given minimum and maximum value (typically between zero and one), as described in <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref>.</p>
<p><disp-formula id="eqn-1">
<label>(1)</label>
<mml:math id="mml-eqn-1" display="block"><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<p>The primary objective of this work is to detect network intrusions effectively. The proposed approach is structured into two main layers: the Base-learner layer and the Combining-Module layer, as depicted in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>. The Base-learner layer consists of multiple classifiers that learn from the data to identify intrusion patterns. In contrast, the Combining-Module layer serves as an aggregator that combines the output from the individual base classifiers, thereby enhancing the overall predictive performance. This layered architecture helps improve intrusion detection accuracy by leveraging the strengths of different classifiers and ensuring a more robust and reliable outcome.</p>
<p>In the first layer, we have selected three base binary classifiers to create a base module for the proposed system: (1) the random forest classifier (R.F.) [<xref ref-type="bibr" rid="ref-31">31</xref>], a method of combining multiple classifiers to tackle complex problems by using several decision trees on different subsets of a dataset and averaging the results to increase prediction accuracy; (2) the decision tree classifier (D.T.) [<xref ref-type="bibr" rid="ref-32">32</xref>], which identifies valuable information from large amounts of random data through predicting values, requiring a training dataset to create a tree and a test dataset to assess the decision tree&#x0027;s accuracy; and finally, (3) the K-NN classifier (K-Nearest Neighbors), a data classification method that calculates the probability of a data point being part of the nearest group, as depicted in the literature.</p>
<p>The ensemble output serves as input to the meta-classifier through the stack generalization process, ultimately yielding the final decision based on logistic regression to predict the probability of a specific class or event (e.g., pass/fail, win/lose). Ensemble learning can be categorized into three main types: bagging, boosting, and stacking. Bagging is the most prevalent method for predicting test outcomes. During the boosting training phase, models are rigorously trained on misclassified data. Stacking, or stacked generalization, is a highly-regarded ensemble technique for enhancing classification performance by combining multiple classifiers, as shown in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>IDS based on stacking (schematic overview)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="IASC_39687-fig-7.tif"/>
</fig>
<p>In contrast to bagging and boosting, stacking comprises two levels: the base learner (level 0) and the meta-learner (level 1) [<xref ref-type="bibr" rid="ref-33">33</xref>]. At the first level, heterogeneous classification models learn from the training data, and their outputs generate a new dataset for the stacking learner. Each new instance in the dataset is assigned to the correct one. Class to be predicted (with a level 0 prediction as an attribute). The meta-learner then utilizes the newly formed dataset to produce the outcome [<xref ref-type="bibr" rid="ref-34">34</xref>], as illustrated in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Stacking Classifiers [<xref ref-type="bibr" rid="ref-36">36</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="IASC_39687-fig-8.tif"/>
</fig>
</sec>
<sec id="s5">
<label>5</label>
<title>Experimental Results and Analysis</title>
<p>This section presents and analyses the suggested IDS findings based on ML methodologies and approaches. After applying seven algorithms to the UNSW-NB15 dataset, we used numerous assessment criteria to determine the efficacy of the proposed IDS system. The following assessment and performance metrics are used to evaluate the proposed IDS&#x2019; performance. The anticipated outcomes are between 0 and 1. Accuracy, Precision, Recall, AUC, F1-Measure, and Mean squared error (MSE) [<xref ref-type="bibr" rid="ref-35">35</xref>] are used the evaluate the proposed approach.
<list list-type="bullet">
<list-item>
<p>Accuracy is the number of correct predictions per classifier.</p></list-item>
</list></p>
<p><disp-formula id="eqn-2">
<label>(2)</label>
<mml:math id="mml-eqn-2" display="block"><mml:mrow><mml:mi mathvariant="italic">A</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">u</mml:mi><mml:mi mathvariant="italic">r</mml:mi><mml:mi mathvariant="italic">a</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">y</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>TN</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mtext>TP</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>TN</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>FP</mml:mtext></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mtext>FN</mml:mtext></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula>
<list list-type="bullet">
<list-item>
<p>Precision: is a measure of genuine positive results derived from all positive findings in the dataset during Classification.</p></list-item>
</list></p>
<p><disp-formula id="eqn-3">
<label>(3)</label>
<mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<list list-type="bullet">
<list-item>
<p>Recall: is the measurement of data points predicted to be positive by all the classifiers.</p></list-item>
</list>
<p><disp-formula id="eqn-4">
<label>(4)</label>
<mml:math id="mml-eqn-4" display="block"><mml:mrow><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula>
<list list-type="bullet">
<list-item>
<p>AUC&#x2014;ROC curve: is a performance metric for classification tasks at various threshold levels. It is defined as the chance that the model rates a random positive sample higher than a random negative sample.</p></list-item>
</list></p>
<p><disp-formula id="eqn-5">
<label>(5)</label>
<mml:math id="mml-eqn-5" display="block"><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<p><disp-formula id="eqn-6">
<label>(6)</label>
<mml:math id="mml-eqn-6" display="block"><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula>
<list list-type="bullet">
<list-item>
<p>F1-Measure: It is the harmonic relationship between Precision and Recall. F1 score is regarded as a superior performance statistic to accuracy. The greater F1-Score rate indicates that the MLmodel is doing better.</p></list-item>
</list></p>
<p><disp-formula id="eqn-7">
<label>(7)</label>
<mml:math id="mml-eqn-7" display="block"><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mi>r</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mi>r</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula>
<list list-type="bullet">
<list-item>
<p>Mean squared error (MSE): represents the expected value of the squared error loss. It is never negative; therefore, numbers close to zero are preferred.</p></list-item>
</list></p>
<p><disp-formula id="eqn-8">
<label>(8)</label>
<mml:math id="mml-eqn-8" display="block"><mml:mrow><mml:mi mathvariant="normal">M</mml:mi><mml:mi mathvariant="normal">S</mml:mi><mml:mi mathvariant="normal">E</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">Y</mml:mi><mml:mi mathvariant="normal">i</mml:mi></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="normal">Y</mml:mi><mml:mi mathvariant="normal">i</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x2227;</mml:mo></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></disp-formula></p>
<p>Collaboratory (Colab) was used for all evaluations, a cloud service based on Jupyter Notebooks for distributing ML teaching and research [<xref ref-type="bibr" rid="ref-37">37</xref>]. The models, pre-processing processes, and measurements were all implemented using the Python programming language in the same Colab environment. The system was running at high speed, and it could be classified the full data quickly so that it could be implemented as a real-world system in the future. We can improve the level of speed if we use collab pro [<xref ref-type="bibr" rid="ref-38">38</xref>] The fastest GPUs are reserved for Colab Pro and Pro&#x002B; customers.</p>
<p>Seven classifiers were examined to assess the performance of the proposed method: Na&#x00EF;ve Bayes (N.B.), linear Support Vector Machine (SVM), SVM with Radial Basis Function (RBF) kernel, k-Nearest Neighbors (KNN), Logistic Regression (L.R.), Decision Tree (D.T.), and Random Forest (R.F.). The proposed method&#x0027;s accuracy was evaluated on the UNSW-NB15 dataset with a full feature space (42 features) for binary Classification. Stratified 10-fold cross-validation was employed to train all models on the training dataset. The 10-fold cross-validation process divides the dataset into ten subsets, with nine used for training the classifiers and one for testing.</p>
<p><xref ref-type="table" rid="table-3">Table 3</xref> presents the accuracy, precision, recall, F1-Score, AUC, and MSE results of the proposed ML algorithms on the training dataset. R.F. achieves the highest accuracy of 96.12%. Linear SVM exhibits the highest precision score of 99.79% but the lowest recall score of 91.21%. While NB has a recall score of 93.41%, its accuracy is considerably lower. Conversely, D.T. demonstrates the highest recall value of 96.38% and minimal variance in its accuracy and recall values, resulting in a higher F1 score than the N.B. model. The R.F. algorithm yields the highest F1 score of 97.18% and the lowest MSE error of 0.0388. To enhance performance, we employed the stacking ensemble method, which comprises two or more base models (level-0 models) and a meta-model that aggregates the base models&#x2019; predictions (level-1 model). The base models include R.F., D.T., and KNN, with the L.R. algorithm as the meta-model. The stacking ensemble achieves 96.16% accuracy.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Performance comparison of ML classifiers on the training dataset</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Method</th>
<th>Accuracy</th>
<th>Precision</th>
<th>Recall</th>
<th>F1-score</th>
<th>AUC</th>
<th>MSE</th>
</tr>
</thead><tbody valign="top">
<tr>
<td>SVM (RBF)</td>
<td>93.60</td>
<td>99.63</td>
<td>91.69</td>
<td>95.49</td>
<td>95.36</td>
<td>0.0640</td>
</tr>
<tr>
<td>Random forest RF</td>
<td>96.12</td>
<td>97.98</td>
<td>96.38</td>
<td>97.18</td>
<td>95.96</td>
<td>0.0388</td>
</tr>
<tr>
<td>Logistic regression LR</td>
<td>93.40</td>
<td>99.13</td>
<td>91.83</td>
<td>95.34</td>
<td>94.79</td>
<td>0.0660</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>93.31</td>
<td>99.79</td>
<td>91.21</td>
<td>95.31</td>
<td>95.33</td>
<td>0.0669</td>
</tr>
<tr>
<td>Na&#x00EF;ve Bayes NB</td>
<td>81.38</td>
<td>78.16</td>
<td>93.41</td>
<td>85.10</td>
<td>79.44</td>
<td>0.1862</td>
</tr>
<tr>
<td>Decision tree DT</td>
<td>95.00</td>
<td>96.27</td>
<td>96.38</td>
<td>96.33</td>
<td>94.23</td>
<td>0.0500</td>
</tr>
<tr>
<td>KNN</td>
<td>93.76</td>
<td>95.90</td>
<td>94.98</td>
<td>95.44</td>
<td>93.02</td>
<td>0.0624</td>
</tr>
<tr>
<td>Stacking ensemble</td>
<td>96.16</td>
<td>97.78</td>
<td>96.62</td>
<td>97.20</td>
<td>95.88</td>
<td>0.0384</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The test dataset results for accuracy, precision, recall, F1-Score, AUC, and MSE of the proposed ML algorithms are displayed in <xref ref-type="table" rid="table-4">Table 4</xref>. R.F. obtains the highest values for accuracy, precision, recall, F1-score, and AUC, with 97.94%, 98.52%, 97.72%, 98.12%, and 97.96%, respectively. D.T. follows with values of 96.74% for accuracy, 96.98% for precision, 97.11% for recall, 97.04% for F1-score, and 96.70% for AUC. KNN ranks third with values of 93.33% for accuracy, 95.56% for precision, 92.17% for recall, 93.84% for F1-score, and 93.46% for AUC. R.F. achieves the best MSE error of 0.0206, followed by D.T. at 0.0326 and KNN at 0.0667. The stacking ensemble attains 97.95% accuracy.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Performance comparison of ML classifiers on the test dataset</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Method</th>
<th>Accuracy</th>
<th>Precision</th>
<th>Recall</th>
<th>F1-score</th>
<th>AUC</th>
<th>MSE</th>
</tr>
</thead><tbody valign="top">
<tr>
<td>SVM (RBF)</td>
<td>93.10</td>
<td>92.65</td>
<td>95.01</td>
<td>93.81</td>
<td>92.89</td>
<td>0.0690</td>
</tr>
<tr>
<td>Random forest RF</td>
<td>97.94</td>
<td>98.52</td>
<td>97.72</td>
<td>98.12</td>
<td>97.96</td>
<td>0.0206</td>
</tr>
<tr>
<td>Logistic regression LR</td>
<td>90.70</td>
<td>91.54</td>
<td>91.58</td>
<td>91.56</td>
<td>90.61</td>
<td>0.0930</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>91.04</td>
<td>92.70</td>
<td>90.89</td>
<td>91.78</td>
<td>91.06</td>
<td>0.0896</td>
</tr>
<tr>
<td>Na&#x00EF;ve Bayes NB</td>
<td>76.43</td>
<td>86.28</td>
<td>68.02</td>
<td>76.07</td>
<td>77.38</td>
<td>0.2357</td>
</tr>
<tr>
<td>Decision tree DT</td>
<td>96.74</td>
<td>96.98</td>
<td>97.11</td>
<td>97.04</td>
<td>96.70</td>
<td>0.0326</td>
</tr>
<tr>
<td>KNN</td>
<td>93.33</td>
<td>95.56</td>
<td>92.17</td>
<td>93.84</td>
<td>93.46</td>
<td>0.0667</td>
</tr>
<tr>
<td>Stacking ensemble</td>
<td>97.95</td>
<td>98.40</td>
<td>97.87</td>
<td>98.13</td>
<td>97.96</td>
<td>0.0205</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>More details can be shown clearly in <xref ref-type="fig" rid="fig-9">Fig. 9</xref>, which discusses the Performance comparison of ML classifiers on the test dataset.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>Performance comparison of ML classifiers on the test dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="IASC_39687-fig-9.tif"/>
</fig>
<p>In conclusion, this study addresses the challenge of identifying the most suitable classifier for a specific classification task, such as intrusion detection in network data. The research compares the performance of various classifiers, including Multilayer Perceptron (MLP), SVM, Decision Trees, and Na&#x00EF;ve Bayes. The findings reveal that employing an ensemble approach, which combines multiple classifiers, mitigates the risk of suboptimal selection compared to relying on a single classifier. The study applies the stacking ensemble technique with base models Random Forest (R.F.), Decision Tree (D.T.), k-Nearest Neighbors (KNN), and the meta-model.</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Conclusion and Future Work</title>
<p>In conclusion, this study has addressed the issue of determining the best classifier for a specific classification task, such as intrusion detection in network data. The study has compared the performance of several classifiers, including Multilayer Perceptron (MLP), SVM (SVM), Decision Trees, and Naive Bayes. The results indicate that an ensemble approach, combining multiple classifiers, reduces the risk of making a poor selection compared to relying on a single classifier. The study has applied the stacking ensemble technique using the base models Random Forest (R.F.), Decision Tree (D.T.), k-Nearest Neighbors (KNN), and the meta-model Logistic Regression (L.R.), and achieved an accuracy of 97.95% in the testing phase. For future research, it is recommended to explore the applicability of this technology in real-world production systems. Additionally, further studies could investigate the combination of different pre-processing techniques and various ensembles to improve the overall performance of IDS.</p>
</sec>
</body>
<back>
<sec><title>Funding Statement</title>
<p>The authors received no specific funding for this study.</p>
</sec>
<sec sec-type="COI-statement"><title>Conflicts of Interest</title>
<p>The authors declare they have no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>R. A.</given-names> <surname>Disha</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Waheed</surname></string-name></person-group>, &#x201C;<article-title>Performance analysis of machine learning models for intrusion detection system using gini impurity-based weighted random forest (GIWRF) feature selection technique</article-title>,&#x201D; <source>Cybersecurity</source>, vol. <volume>5</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>22</lpage>, <year>2022</year>. <pub-id pub-id-type="doi">10.1186/s42400-021-00103-8</pub-id></mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. M.</given-names> <surname>Najafabadi</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Villanustre</surname></string-name>, <string-name><given-names>T. M.</given-names> <surname>Khoshgoftaar</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Seliya</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Wald</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Deep learning applications and challenges in big data analytics</article-title>,&#x201D; <source>Journal of Big Data</source>, vol. <volume>2</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>21</lpage>, <year>2015</year>. <pub-id pub-id-type="doi">10.1186/s40537-014-0007-7</pub-id></mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Dong</surname></string-name> and <string-name><given-names>X.</given-names> <surname>Wang</surname></string-name></person-group>, &#x201C;<article-title>Comparison deep learning method to traditional methods using for network intrusion detection</article-title>,&#x201D; in <conf-name>8th IEEE Int. Conf. on Communication Software and Networks (ICCSN)</conf-name>, <publisher-loc>Beijing, China</publisher-loc>, pp. <fpage>581</fpage>&#x2013;<lpage>585</lpage>, <year>2016</year>. </mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>J. P.</given-names> <surname>Anderson</surname></string-name></person-group>, &#x201C;<article-title>Computer security threat monitoring and surveillance</article-title>,&#x201D; <comment>Technical Report, James P. Anderson Company</comment>, <year>1980</year>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Debar</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Dacier</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Wespi</surname></string-name></person-group>, &#x201C;<article-title>Towards a taxonomy of intrusion-detection systems</article-title>,&#x201D; <source>Computer Networks</source>, vol. <volume>31</volume>, no. <issue>8</issue>, pp. <fpage>805</fpage>&#x2013;<lpage>822</lpage>, <year>1999</year>. <pub-id pub-id-type="doi">10.1016/S1389-1286(98)00017-6</pub-id></mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Mukkamala</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Janoski</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Sung</surname></string-name></person-group>, &#x201C;<article-title>Intrusion detection using neural networks and support vector machines</article-title>,&#x201D; in <conf-name>2002 Int. Joint Conf. on Neural Networks. IJCNN&#x2019;02 (Cat. No. 02CH37290)</conf-name>, <publisher-loc>Honolulu, HI, USA</publisher-loc>, vol. <volume>15</volume>, pp. <fpage>1702</fpage>&#x2013;<lpage>1707</lpage>, <year>2002</year>. </mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Lew</surname></string-name>, <string-name><given-names>D. A.</given-names> <surname>Shah</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Pati</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Cattell</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Zhang</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Analyzing machine learning workloads using a detailed GPU simulator</article-title>,&#x201D; in <conf-name>IEEE Int. Symp. on Performance Analysis of Systems and Software (ISPASS)</conf-name>, <publisher-loc>Madison, WI, USA</publisher-loc>, pp. <fpage>151</fpage>&#x2013;<lpage>152</lpage>, <year>2019</year>. </mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Abirami</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Yash</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Singh</surname></string-name></person-group>, &#x201C;<chapter-title>Building an ensemble learning based algorithm for improving intrusion detection system</chapter-title>,&#x201D; In: <person-group person-group-type="editor"><string-name><given-names>M.</given-names> <surname>Abirami</surname></string-name></person-group> (Ed.), <source>Artificial Intelligence and Evolutionary Computations in Engineering Systems</source>, pp. <fpage>635</fpage>&#x2013;<lpage>649</lpage>, <publisher-name>Singapore: Springer</publisher-name>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Khraisat</surname></string-name>, <string-name><given-names>I.</given-names> <surname>Gondal</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Vamplew</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Kamruzzaman</surname></string-name></person-group>, &#x201C;<article-title>Survey of intrusion detection systems: Techniques, datasets and challenges</article-title>,&#x201D; <source>Cybersecurity</source>, vol. <volume>2</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>22</lpage>, <year>2019</year>. <pub-id pub-id-type="doi">10.1186/s42400-019-0038-7</pub-id></mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J. H.</given-names> <surname>Liao</surname></string-name>, <string-name><given-names>C. H. R.</given-names> <surname>Lin</surname></string-name>, <string-name><given-names>Y. C.</given-names> <surname>Lin</surname></string-name> and <string-name><given-names>K. Y.</given-names> <surname>Tung</surname></string-name></person-group>, &#x201C;<article-title>Intrusion detection system: A comprehensive review</article-title>,&#x201D; <source>Journal of Network and Computer Applications</source>, vol. <volume>36</volume>, no. <issue>1</issue>, pp. <fpage>16</fpage>&#x2013;<lpage>24</lpage>, <year>2013</year>. <pub-id pub-id-type="doi">10.1016/j.jnca.2012.09.004</pub-id></mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Rajadurai</surname></string-name> and <string-name><given-names>U. D.</given-names> <surname>Gandhi</surname></string-name></person-group>, &#x201C;<article-title>A stacked ensemble learning model for intrusion detection in wireless network</article-title>,&#x201D; <source>Neural Computing and Applications</source>, pp. <fpage>1</fpage>&#x2013;<lpage>9</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X.</given-names> <surname>Gao</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Shan</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Hu</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Niu</surname></string-name> and <string-name><given-names>Z.</given-names> <surname>Liu</surname></string-name></person-group>, &#x201C;<article-title>An adaptive ensemble machine learning model for intrusion detection</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>7</volume>, pp. <fpage>82512</fpage>&#x2013;<lpage>82521</lpage>, <year>2019</year>. <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2923640</pub-id></mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Lutkevich</surname></string-name></person-group>, <source>What is an intrusion detection system (IDS)?</source> <comment>USA: Techtarget</comment>, <year>2020</year>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.techtarget.com/searchsecurity/definition/intrusion-detection-system">https://www.techtarget.com/searchsecurity/definition/intrusion-detection-system</ext-link></mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X.</given-names> <surname>Larriva-Novo</surname></string-name>, <string-name><given-names>C.</given-names> <surname>S&#x00E1;nchez-Zas</surname></string-name>, <string-name><given-names>V. A.</given-names> <surname>Villagr&#x00E1;</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Vega-Barbas</surname></string-name> and <string-name><given-names>D.</given-names> <surname>Rivera</surname></string-name></person-group>, &#x201C;<article-title>An approach for the application of a dynamic multi-class classifier for network intrusion detection systems</article-title>,&#x201D; <source>Electronics</source>, vol. <volume>9</volume>, no. <issue>11</issue>, pp. <fpage>1759</fpage>, <year>2020</year>. <pub-id pub-id-type="doi">10.3390/electronics9111759</pub-id></mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Slay</surname></string-name></person-group>, &#x201C;<article-title>UNSW-NB15: A comprehensive data set for network intrusion detection systems (UNSW-NB15 network data set)</article-title>,&#x201D; in <conf-name>IEEE Military Communications and Information Systems Conf. (MilCIS)</conf-name>, <publisher-loc>Canberra, ACT, Australia</publisher-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>6</lpage>, <year>2015</year>. </mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Khraisat</surname></string-name>, <string-name><given-names>I.</given-names> <surname>Gondal</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Vamplew</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Kamruzzaman</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Alazab</surname></string-name></person-group>, &#x201C;<article-title>Hybrid intrusion detection system based on the stacking ensemble of c5 decision tree classifier and one class support vector machine</article-title>,&#x201D; <source>Electronics</source>, vol. <volume>9</volume>, no. <issue>1</issue>, pp. <fpage>173</fpage>, <year>2020</year>. <pub-id pub-id-type="doi">10.3390/electronics9010173</pub-id></mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Rababah</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Srivastava</surname></string-name></person-group>, &#x201C;<article-title>Hybrid model for intrusion detection systems</article-title>,&#x201D; <comment><italic>ArXiv Preprint arXiv:2003.08585</italic></comment>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Jing</surname></string-name> and <string-name><given-names>H. -B.</given-names> <surname>Chen</surname></string-name></person-group>, &#x201C;<article-title>SVM based network intrusion detection for the UNSW-NB15 dataset</article-title>,&#x201D; in <conf-name>IEEE 13th Int. Conf. on ASIC (ASICON)</conf-name>, <publisher-loc>Chongqing, China</publisher-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>4</lpage>, <year>2019</year>. </mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>I.</given-names> <surname>Ahmad</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Basheri</surname></string-name>, <string-name><given-names>M. J.</given-names> <surname>Iqbal</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Rahim</surname></string-name></person-group>, &#x201C;<article-title>Performance comparison of support vector machine, random forest, and extreme learning machine for intrusion detection</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>6</volume>, pp. <fpage>33789</fpage>&#x2013;<lpage>33795</lpage>, <year>2018</year>. <pub-id pub-id-type="doi">10.1109/ACCESS.2018.2841987</pub-id></mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Ring</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Wunderlich</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Scheuring</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Landes</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Hotho</surname></string-name></person-group>, &#x201C;<article-title>A survey of network-based intrusion detection data sets</article-title>,&#x201D; <source>Computers &#x0026; Security</source>, vol. <volume>86</volume>, no. <issue>1</issue>, pp. <fpage>147</fpage>&#x2013;<lpage>167</lpage>, <year>2019</year>. <pub-id pub-id-type="doi">10.1016/j.cose.2019.06.005</pub-id></mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. M. H.</given-names> <surname>Bamakan</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Yingjie</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Shi</surname></string-name></person-group>, &#x201C;<article-title>An effective intrusion detection framework based on mclp/svm optimized by time-varying chaos particle swarm optimization</article-title>,&#x201D; <source>Neurocomputing</source>, vol. <volume>199</volume>, no. <issue>10</issue>, pp. <fpage>90</fpage>&#x2013;<lpage>102</lpage>, <year>2016</year>. <pub-id pub-id-type="doi">10.1016/j.neucom.2016.03.031</pub-id></mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Z. -H.</given-names> <surname>Pang</surname></string-name>, <string-name><given-names>G. -P.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Zhou</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Hou</surname></string-name> and <string-name><given-names>D.</given-names> <surname>Sun</surname></string-name></person-group>, &#x201C;<article-title>Two-channel false data injection attacks against output tracking control of networked systems</article-title>,&#x201D; <source>IEEE Transactions on Industrial Electronics</source>, vol. <volume>63</volume>, no. <issue>5</issue>, pp. <fpage>3242</fpage>&#x2013;<lpage>3251</lpage>, <year>2016</year>.<pub-id pub-id-type="doi">10.1109/TIE.2016.2535119</pub-id></mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. S. S.</given-names> <surname>Sindhu</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Geetha</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Kannan</surname></string-name></person-group>, &#x201C;<article-title>Decision tree based light weight intrusion detection using a wrapper approach</article-title>,&#x201D; <source>Expert Systems with Applications</source>, vol. <volume>39</volume>, no. <issue>1</issue>, pp. <fpage>129</fpage>&#x2013;<lpage>141</lpage>, <year>2012</year>. <pub-id pub-id-type="doi">10.1016/j.eswa.2011.06.013</pub-id></mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Lee</surname></string-name>, <string-name><given-names>S. J.</given-names> <surname>Stolfo</surname></string-name> and <string-name><given-names>K. W.</given-names> <surname>Mok</surname></string-name></person-group>, &#x201C;<article-title>A data mining framework for building intrusion detection models</article-title>,&#x201D; in <conf-name>IEEE Symp. on Security and Privacy (Cat. No. 99CB36344)</conf-name>, <publisher-loc>Oakland, CA, USA</publisher-loc>, pp. <fpage>120</fpage>&#x2013;<lpage>132</lpage>, <year>1999</year>. </mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. G.</given-names> <surname>Raman</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Somu</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Kirthivasan</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Liscano</surname></string-name> and <string-name><given-names>V. S.</given-names> <surname>Sriram</surname></string-name></person-group>, &#x201C;<article-title>An efficient intrusion detection system based on hypergraph-genetic algorithm for parameter optimization and feature selection in support vector machine</article-title>,&#x201D; <source>Knowledge-Based Systems</source>, vol. <volume>134</volume>, no. <issue>5</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>12</lpage>, <year>2017</year>. <pub-id pub-id-type="doi">10.1016/j.knosys.2017.07.005</pub-id></mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. G.</given-names> <surname>Raman</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Somu</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Jagarapu</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Manghnani</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Selvam</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>An efficient intrusion detection technique based on support vector machine and improved binary gravitational search algorithm</article-title>,&#x201D; <source>Artificial Intelligence Review</source>, vol. <volume>53</volume>, no. <issue>5</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>32</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Rashid</surname></string-name>, <string-name><given-names>M. J.</given-names> <surname>Siddique</surname></string-name> and <string-name><given-names>S. M.</given-names> <surname>Ahmed</surname></string-name></person-group>, &#x201C;<article-title>Machine and deep learning based comparative analysis using hybrid approaches for intrusion detection system</article-title>,&#x201D; in <conf-name>3rd Int. Conf. on Advancements in Computational Sciences (ICACS)</conf-name>, <publisher-loc>Lahore, Pakistan</publisher-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>9</lpage>, <year>2020</year>. </mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Yang</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Qin</surname></string-name> and <string-name><given-names>L.</given-names> <surname>Ye</surname></string-name></person-group>, &#x201C;<article-title>Combined wireless network intrusion detection model based on deep learning</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>7</volume>, pp. <fpage>82624</fpage>&#x2013;<lpage>82632</lpage>, <year>2019</year>. <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2923814</pub-id></mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>R. K. S.</given-names> <surname>Gautam</surname></string-name> and <string-name><given-names>E. A.</given-names> <surname>Doegar</surname></string-name></person-group>, &#x201C;<article-title>An ensemble approach for intrusion detection system using machine learning algorithms</article-title>,&#x201D; in <conf-name>2018 8th Int. Conf. on Cloud Computing, Data Science &#x0026; Engineering (Confluence)</conf-name>, <publisher-loc>Noida, India</publisher-loc>, pp. <fpage>14</fpage>&#x2013;<lpage>15</lpage>, <year>2018</year>. </mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>Z.</given-names> <surname>Zoghi</surname></string-name> and <string-name><given-names>G.</given-names> <surname>Serpen</surname></string-name></person-group>, &#x201C;<article-title>UNSW-NB15 computer security dataset: Analysis through visualization</article-title>,&#x201D; <comment><italic>arXiv preprint arXiv:2101.05067</italic></comment>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. S.</given-names> <surname>Nikam</surname></string-name></person-group>, &#x201C;<article-title>A comparative study of classification techniques in data mining algorithms</article-title>,&#x201D; <source>Oriental Journal of Computer Science and Technology</source>, vol. <volume>8</volume>, no. <issue>1</issue>, pp. <fpage>13</fpage>&#x2013;<lpage>19</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Kaur</surname></string-name> and <string-name><given-names>H.</given-names> <surname>Kaur</surname></string-name></person-group>, &#x201C;<article-title>Review of decision tree data mining algorithms: Cart and c4. 5</article-title>,&#x201D; <source>International Journal of Advanced Research in Computer Science</source>, vol. <volume>8</volume>, no. <issue>4</issue>, pp. <fpage>436</fpage>&#x2013;<lpage>439</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>I.</given-names> <surname>Syarif</surname></string-name>, <string-name><given-names>E.</given-names> <surname>Zaluska</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Prugel-Bennett</surname></string-name> and <string-name><given-names>G.</given-names> <surname>Wills</surname></string-name></person-group>, &#x201C;<article-title>Application of bagging, boosting and stacking to intrusion detection</article-title>,&#x201D; in <conf-name>Int. Workshop on Machine Learning and Data Mining in Pattern Recognition</conf-name>, <publisher-loc>Berlin, Germany</publisher-loc>, pp. <fpage>593</fpage>&#x2013;<lpage>602</lpage>, <year>2012</year>. </mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Graczyk</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Lasota</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Trawi&#x0144;ski</surname></string-name> and <string-name><given-names>K.</given-names> <surname>Trawi&#x0144;ski</surname></string-name></person-group>, &#x201C;<article-title>Comparison of bagging, boosting and stacking ensembles applied to real estate appraisal</article-title>,&#x201D; in <conf-name>Asian Conf. on Intelligent Information and Database Systems</conf-name>, <publisher-loc>Hue, Vietnam</publisher-loc>, pp. <fpage>340</fpage>&#x2013;<lpage>350</lpage>, <year>2010</year>. </mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><given-names>C. M.</given-names> <surname>Bishop</surname></string-name> and <string-name><given-names>N. M.</given-names> <surname>Nasrabadi</surname></string-name></person-group>, <source>Pattern Recognition and Machine Learning (No. 4)</source>. <publisher-name>New York, USA: Springer</publisher-name>, <year>2006</year>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Ceballos</surname></string-name></person-group>, <source>Stacking classifiers for higher predictive performance</source>. <comment>USA: Medium</comment>, <year>2022</year>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://towardsdatascience.com/stacking-classifiers-for-higher-predictive-performance-566f963e4840">https://towardsdatascience.com/stacking-classifiers-for-higher-predictive-performance-566f963e4840</ext-link></mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Introne</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Laubacher</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Olson</surname></string-name> and <string-name><given-names>T.</given-names> <surname>Malone</surname></string-name></person-group>, &#x201C;<article-title>The climate colab: Large scale model-based collaborative planning</article-title>,&#x201D; in <conf-name>Int. Conf. on Collaboration Technologies and Systems (CTS)</conf-name>, <publisher-loc>Philadelphia, PA, USA</publisher-loc>, pp. <fpage>40</fpage>&#x2013;<lpage>47</lpage>, <year>2011</year>. </mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><given-names>E.</given-names> <surname>Bisong</surname></string-name></person-group>, &#x201C;<chapter-title>Google colaboratory</chapter-title>,&#x201D; in <source>Building Machine Learning and Deep Learning Models on Google Cloud Platform: A Comprehensive Guide for Beginners</source>, <publisher-loc>Berkeley, CA</publisher-loc>: <publisher-name>Apress</publisher-name>, pp. <fpage>59</fpage>&#x2013;<lpage>64</lpage>, <year>2019</year>.</mixed-citation></ref>
</ref-list>
</back>
</article>