<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CSSE</journal-id>
<journal-id journal-id-type="nlm-ta">CSSE</journal-id>
<journal-id journal-id-type="publisher-id">CSSE</journal-id>
<journal-title-group>
<journal-title>Computer Systems Science &#x0026; Engineering</journal-title>
</journal-title-group>
<issn pub-type="ppub">0267-6192</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">18086</article-id>
<article-id pub-id-type="doi">10.32604/csse.2021.018086</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Semisupervised Encrypted Traffic Identification Based on Auxiliary Classification Generative Adversarial Network</article-title><alt-title alt-title-type="left-running-head">Semisupervised Encrypted Traffic Identification Based on Auxiliary Classification Generative Adversarial Network</alt-title><alt-title alt-title-type="right-running-head">Semisupervised Encrypted Traffic Identification Based on Auxiliary Classification Generative Adversarial Network</alt-title>
</title-group>
<contrib-group content-type="authors">
<contrib id="author-1" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Mao</surname><given-names>Jiaming</given-names></name>
<xref ref-type="aff" rid="aff-1">1</xref><email>513222462@qq.com</email>
</contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Zhang</surname><given-names>Mingming</given-names></name>
<xref ref-type="aff" rid="aff-1">1</xref>
</contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Chen</surname><given-names>Mu</given-names></name>
<xref ref-type="aff" rid="aff-2">2</xref>
</contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Chen</surname><given-names>Lu</given-names></name>
<xref ref-type="aff" rid="aff-2">2</xref>
</contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Xia</surname><given-names>Fei</given-names></name>
<xref ref-type="aff" rid="aff-1">1</xref>
</contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Fan</surname><given-names>Lei</given-names></name>
<xref ref-type="aff" rid="aff-1">1</xref>
</contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Wang</surname><given-names>ZiXuan</given-names></name>
<xref ref-type="aff" rid="aff-3">3</xref>
</contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>Zhao</surname><given-names>Wenbing</given-names></name>
<xref ref-type="aff" rid="aff-4">4</xref>
</contrib>
<aff id="aff-1"><label>1</label><institution>State Grid Jiangsu Electric Power Co., Ltd. Information and Telecommunication Branch</institution>, <addr-line>NanJing, 210003</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>Institute of Information and Communication, Global Energy Interconnection Research Institute, State Grid Key Laboratory of Information and Network Security</institution>, <addr-line>Nanjing, 210003</addr-line>, <country>China</country></aff>
<aff id="aff-3"><label>3</label><institution>Engineering Research Center of Post Big Data Technology and Application of Jiangsu Province, Research and Development Center of Post Industry Technology of the State Posts Bureau (Internet of Things Technology), Broadband Wireless Communication Technology Engineering Research Center of the Ministry of Education, Nanjing University of Posts and Telecommunications</institution>, <addr-line>Nanjing, 210003</addr-line>, <country>China</country></aff>
<aff id="aff-4"><label>4</label><institution>Department of Electrical Engineering and Computer Science, Cleveland State University</institution>, <addr-line>Cleveland, 44115</addr-line>, <country>USA</country></aff>
</contrib-group><author-notes><corresp id="cor1">&#x002A;Corresponding Author: Jiaming Mao. Email: <email>513222462@qq.com</email></corresp></author-notes>
<pub-date pub-type="epub" date-type="pub" iso-8601-date="2021-07-29"><day>29</day>
<month>07</month>
<year>2021</year></pub-date>
<volume>39</volume>
<issue>3</issue>
<fpage>373</fpage>
<lpage>390</lpage>
<history>
<date date-type="received"><day>24</day><month>2</month><year>2021</year></date>
<date date-type="accepted"><day>21</day><month>4</month><year>2021</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2021 Mao et al.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Mao et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CSSE_18086.pdf"></self-uri>
<abstract>
<p>The rapidly increasing popularity of mobile devices has changed the methods with which people access various network services and increased network traffic markedly. Over the past few decades, network traffic identification has been a research hotspot in the field of network management and security monitoring. However, as more network services use encryption technology, network traffic identification faces many challenges. Although classic machine learning methods can solve many problems that cannot be solved by port- and payload-based methods, manually extract features that are frequently updated is time-consuming and labor-intensive. Deep learning has good automatic feature learning capabilities and is an ideal method for network traffic identification, particularly encrypted traffic identification; Existing recognition methods based on deep learning primarily use supervised learning methods and rely on many labeled samples. However, in real scenarios, labeled samples are often difficult to obtain. This paper adjusts the structure of the auxiliary classification generation adversarial network (ACGAN) so that it can use unlabeled samples for training, and use the wasserstein distance instead of the original cross entropy as the loss function to achieve semisupervised learning. Experimental results show that the identification accuracy of ISCX and USTC data sets using the proposed method yields markedly better performance when the number of labeled samples is small compared to that of convolutional neural network (CNN) based classifier.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Encrypted traffic recognition</kwd>
<kwd>deep learning</kwd>
<kwd>generative adversarial network</kwd>
<kwd>traffic classification</kwd>
<kwd>semisupervised learning</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Traffic classification and identification can be used to improve network management and security monitoring to improve service quality and provide a foundation for network design and planning. Network traffic identification has been studied in depth. Based on the classification methods used, network traffic identification can be divided into methods based on host attributes, payload, machine learning and deep learning. As early as 1995, scholars used traditional identification methods based on service host attributes to identify network traffic [<xref ref-type="bibr" rid="ref-1">1</xref>]. As more programs were developed to use dynamically allocated port numbers to camouflage traffic, traditional methods began to fail quickly. Academia then began to study the method of mining the flow characteristics of applications through the flow payload to classify applications [<xref ref-type="bibr" rid="ref-2">2</xref>,<xref ref-type="bibr" rid="ref-3">3</xref>]. The current identification method based on payload can manage most identification problems in nonencrypted traffic scenarios but cannot identify encrypted traffic.</p>
<p>With the enhancement of user security awareness and the wide application of SSL, SSH, VPN and other technologies, the proportion of encrypted traffic in network transmission is increasing. Traditional methods have been unable to accurately identify application traffic. Many scholars use machine learning to manage the problem of encrypted traffic identification. Methods based on machine learning do not need to analyze the specific structure of traffic packets, and their identification algorithms will automatically extract traffic characteristics to form a classifier [<xref ref-type="bibr" rid="ref-4">4</xref>&#x2013;<xref ref-type="bibr" rid="ref-6">6</xref>]. However, the feature design of machine learning must rely on the experience of experts, and the constantly changing characteristics of encrypted traffic make this work time-consuming and labor-intensive. With increasing demand for encrypted traffic identification, the high analysis and labor costs of traditional machine learning have gradually become prohibitive.</p>
<p>Compared to machine learning, deep learning can better express the essential characteristics of data, which is attractive in many applications. [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-8">8</xref>] For example, reference Wang et al. [<xref ref-type="bibr" rid="ref-9">9</xref>] uses a data packet header and payload as input and uses convolutional neural network (CNN) and long short-term memory network (LSTM) for traffic identification. References Aceto et al. [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-11">11</xref>] use the first N bytes of the payload and original data; certain features in the first 20 packets before interactive communication is used as input; and the algorithm uses multilayer perceptron (MLP) for traffic identification. Reference Lopez-Martin et al. [<xref ref-type="bibr" rid="ref-12">12</xref>] combines multiple neural networks for IoT traffic identification, and reference H&#x00F6;chst et al. [<xref ref-type="bibr" rid="ref-13">13</xref>] uses an autoencoder (SAE) to determine web browsing interactions, game downloads, online playback uploads and other actions in network traffic. Most existing deep learning-based traffic recognition methods are based on supervised learning methods and rely on a large amount of labeled data. Conversely, unlabeled data is relatively easy to obtain. How to combine a large amount of unlabeled traffic data with a small amount of labeled traffic data to complete the classification task in a semisupervised way and alleviate the dependence of many of labeled data sets has important research significance.</p>
<p>To describe semisupervised learning, this paper proposes a semisupervised encrypted traffic recognition method based on the auxiliary classification generation adversarial network (SACGAN). This method implements semisupervised learning with the help of a small amount of encrypted traffic data from real applications, the data generated by the generated adversarial network (GAN), and unlabeled real traffic data. In the case of few labeled data samples, better recognition results can be obtained than supervised learning methods under the same conditions. The main contribution of this article is summarized as follows:<list list-type="alpha-lower"><list-item>
<p>The network structure of the ACGAN is modified, and the loss function of the generator and discriminator is modified to make it possible to use unlabeled samples for semisupervised learning.</p></list-item><list-item>
<p>The improved network structure is used for encrypted traffic identification, and a semisupervised encrypted traffic identification scheme based on the ACGAN is designed.</p></list-item><list-item>
<p>Using the proposed method and the CNN classification method in the paper [<xref ref-type="bibr" rid="ref-14">14</xref>], a classification experiment was performed on the ISCX and USTC data sets. The results show that under the same conditions, the proposed method is more accurate when there are fewer labeled samples.</p></list-item></list></p>
<p>The organizational structure of this paper is as follows. The first chapter is an introduction, which introduces the current research status in the field of encryption traffic identification and the purpose and significance of this paper&#x2019;s work. The second chapter discusses related studies and introduces the research status of the encrypted traffic identification and semisupervised encrypted traffic identification using GAN. The third chapter introduces the proposed methods, including the network structure, loss function and training method of the semisupervised auxiliary classification generating adversarial network (SACGAN). In chapter four, we introduce the proposed experiments, and perform classification experiments on ISCX and USTC data sets. Chapter five summarizes the article and proposes future research.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Work</title>
<p>Since Goodfellow proposed GAN in 2014 [<xref ref-type="bibr" rid="ref-15">15</xref>], GAN has been considered a promising technology. In recent years, GAN has demonstrated advantages in image, sound, and text generation [<xref ref-type="bibr" rid="ref-13">13</xref>&#x2013;<xref ref-type="bibr" rid="ref-20">20</xref>]. Due to similarities between traffic data, text and sentences, scholars consider using GAN to generate traffic data. GAN is often used to generate adversarial attack traffic to spoof detection systems [<xref ref-type="bibr" rid="ref-21">21</xref>&#x2013;<xref ref-type="bibr" rid="ref-23">23</xref>]. Because data consistent with the real data distribution can be generated through antagonistic learning, GAN can also be used to balance the management of traffic data sets. The authors of reference [<xref ref-type="bibr" rid="ref-24">24</xref>] use an unsupervised learning method called auxiliary classi-fier GAN (AC-GAN) to generate comprehensive traffic samples to balance primary and secondary classes on NIMS, a well-known traffic data set that contains only SSH and non-SSH classes. Douzas et al. [<xref ref-type="bibr" rid="ref-25">25</xref>] used a conditional generation adversarial network (CGAN) to obtain the true distribution of the small sample by adding conditional in-formation to the GAN. A trained generator is then used to generate flow and perform sample balancing. However, instability and pattern collapse problems often occur in this model during training [<xref ref-type="bibr" rid="ref-26">26</xref>&#x2013;<xref ref-type="bibr" rid="ref-28">28</xref>]. Similarly, Zheng et al. [<xref ref-type="bibr" rid="ref-29">29</xref>] proposed CWGAN-GP as a new oversampling method, which can learn from real data distribution based on the global information of the data set thus improving the low recognition rate of mi-nority groups in the imbalanced data set and concurrently solving the problem of mode collapse.</p>
<p>Currently, the semisupervised framework based on deep learning is primarily used in the fields of computer vision and natural language processing [<xref ref-type="bibr" rid="ref-30">30</xref>,<xref ref-type="bibr" rid="ref-31">31</xref>]. Existing research on semisupervised deep learning architecture is divided into two categories: (1) methods of unsupervised pretraining followed by supervised adjustment, (2) methods of unsupervised and supervised training simultaneously. Recently, Rezaei et al. [<xref ref-type="bibr" rid="ref-32">32</xref>] reduced the size of the tagged data set required to train deep learning classifiers using a semisupervised learning approach based on the concept of transfer learning. They pretrained a CNN model with large unlabeled data sets and then transfer the learned weights to the new model using labeled data sets to train it and achieved good results. However, semisupervision based on transfer learning is differ-ent from the method proposed in this paper. Conversely, the proposed approach has an end-to-end advantage. Although there are various methods for semisupervised learn-ing using GAN, there are relatively few studies on semisupervised recognition of en-crypted traffic using GAN. Reference Salimans et al. [<xref ref-type="bibr" rid="ref-33">33</xref>] proposes to produce real data from the cat-egory and regard the generated data as the category. They assume that half of the data comes from the true distribution and define supervised and unsupervised losses. The discriminator minimizes supervised and unsupervised losses during training, while the generator is the opposite. BadGAN [<xref ref-type="bibr" rid="ref-34">34</xref>] theoretically showed the feasibility of using a complementary generator to alternately train the model by minimizing the KL diver-gence between the distribution and target, and maximizing the conditional entropy of the discriminator. CatGAN [<xref ref-type="bibr" rid="ref-35">35</xref>] replaces the discriminator in a GAN from binary clas-sification to multiclassification, takes the cross entropy between the distribution of la-beled samples and the conditional distribution of unlabeled samples as the objective function, and minimizes the cross entropy of labeled data during training. The cross entropy and conditional entropy of real data are used to optimize the discriminator, and the cross entropy of real data categories and conditional entropy of generated data are maximized for semisupervised classification. Although these methods use the un-supervised data generated by a GAN to train the discriminator supervised and de-scribe semisupervision, many unlabeled samples are not used in the training process.</p>
<p>Iliyasu et al. [<xref ref-type="bibr" rid="ref-36">36</xref>] proposed a semisupervised learning method based on a deep convolutional Generative Adversarial Network (DCGAN). This network uses samples generated by the generator and unlabeled data to improve performance on a few labeled samples. The performance of the trained classifier reduces the difficulty of data set collection and labeling. Their method can use few labeled samples (only 10% of the data set) to achieve 89% and 78% accuracy on the QUIC and ISCX VPN-NonVPN data sets, respectively. We use the auxiliary classification generation adversarial net-work (ACGAN) to generate traffic samples by using class labels, and modify the loss function of the generator, which also achieved good results on ISCX and USTC data sets.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Proposed Approach</title>
<sec id="s3_1">
<label>3.1</label>
<title>Overall Structure</title>
<p>Based on the semisupervised classification model idea, we apply an ACGAN [<xref ref-type="bibr" rid="ref-37">37</xref>], which is widely used in image and video fields, to the semisupervised traffic classification field and include the following three steps: data preprocessing, model training, and semisupervised traffic recognition. When a traffic classification task is to be performed, the following steps are followed:<list list-type="alpha-lower"><list-item>
<p>First, the original data is preprocessed, and the traffic packet data is filtered, truncated/zero-filled, and standardized to form a packet byte vector (PBV).</p></list-item><list-item>
<p>Then the SACGAN is constructed based on the method described in this article. The standardized PBV is sent to the semisupervised traffic classification model, and the SACGAN are trained using the labeled and unlabeled real data.</p></list-item><list-item>
<p>Iteratively update the network to stabilize the SACGAN network and calculate the classification results. <xref ref-type="fig" rid="fig-1">Fig. 1</xref> shows the overall flow of the proposed SACGAN for traffic identification.</p></list-item></list></p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Overall SACGAN process</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-1.png"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Improved Semisupervised Auxiliary Classification Generative Adversarial Network (SACGAN)</title>
<sec id="s3_2_1">
<label>3.2.1</label>
<title>Network Structure</title>
<p><xref ref-type="fig" rid="fig-2">Fig. 2</xref> shows the network structure of SACGAN. It consists of two parts: generator and discriminator. The generator receives the combined vector of noise and tags to generate traffic data. The discriminator receives real labeled data, real unlabeled data and data generated by the generator to output true and false discrimination and class labels.</p>
<p>During design, the network structure of the DCGAN [<xref ref-type="bibr" rid="ref-38">38</xref>], WGAN [<xref ref-type="bibr" rid="ref-39">39</xref>] and the semisupervised model [<xref ref-type="bibr" rid="ref-40">40</xref>] are referenced as follows:<list list-type="alpha-lower"><list-item>
<p>We use a convolutional layer with strides to replace the pooling layer, use con-volution to replace the pooling layer of the discriminator network, and use de-convolution to replace the pooling layer of the generated model.</p></list-item><list-item>
<p>LeakyReLU [<xref ref-type="bibr" rid="ref-41">41</xref>] is used for activation in both the generator and discriminator, and the generator output layer uses the tanh function.</p></list-item><list-item>
<p>After updating the parameters of the discriminator, the absolute value is truncated to not exceed a fixed constant C.</p></list-item></list></p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Network structure of the SACGAN</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-2.png"/>
</fig>
<p>During implementation, the discriminator is composed of a CNN network for feature extraction and two MLP networks for classification and true-false discrimina-tion. Based on reference Salimans et al. [<xref ref-type="bibr" rid="ref-40">40</xref>], the output layer of the discriminator is built using a stacked model with shared weights. The two MLP networks multiplex the output of the CNN network. In Keras, custom functions are applied from the Lambda layer to the input layer of the network.</p>
</sec>
<sec id="s3_2_2">
<label>3.2.2</label>
<title>Discriminator Structure</title>
<p>a. Loss function</p>
<p>The loss function of the SACGAN&#x2019;s discriminator has two parts (classification loss and adversarial loss), as shown in formula (1). Formula 2) represents the adversarial loss, (i.e., the unlabeled loss), and formula (3) represents the classification loss, (i.e., the labeled loss):</p>
<p><disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>D</mml:mi></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x002B;</mml:mo><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:math>
</disp-formula></p>
<p>Define <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math>
</inline-formula> as the probability distribution of real data (including labeled data and unlabeled data). <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:math>
</inline-formula> is defined as the probability distribution of generated data. For the adversarial loss of the discriminator D, the Wasserstein distance is used instead of the original log-likelihood function, which represents the Earth-Mover (EM) distance from <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mspace width="thickmathspace"></mml:mspace></mml:mrow></mml:math>
</inline-formula> to <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:math>
</inline-formula> [<xref ref-type="bibr" rid="ref-39">39</xref>]:</p>
<p><disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2"><mml:mi>W</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>K</mml:mi></mml:mfrac></mml:mrow><mml:munder><mml:mrow><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">&#x2225;</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:msub><mml:mo>&#x2225;</mml:mo><mml:mi>L</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2264;</mml:mo><mml:mi>K</mml:mi></mml:mrow></mml:munder><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">]</mml:mo></mml:math>
</disp-formula></p>
<p><disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mo stretchy="false">&#x2225;</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:msub><mml:mo>&#x2225;</mml:mo><mml:mi>L</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2264;</mml:mo><mml:mi>K</mml:mi></mml:math>
</disp-formula></p>
<p>Formula (3) represents Lipschitz continuous, which means that an additional restriction is imposed on the function <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math>
</inline-formula>, requiring a constant <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mi>K</mml:mi><mml:mo>&#x2265;</mml:mo><mml:mn>0</mml:mn></mml:math>
</inline-formula> so that any two elements <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:math>
</inline-formula> and <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:math>
</inline-formula> in the domain satisfy formula (4)</p>
<p><disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>&#x2264;</mml:mo><mml:mi>K</mml:mi><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<p>In particular, we can use a set of parameters <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mi>w</mml:mi></mml:math>
</inline-formula> to define a series of possible functions <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math>
</inline-formula>. At this time, formula (2) can be approximated into the following form:</p>
<p><disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5"><mml:mi>K</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>W</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2248;</mml:mo><mml:munder><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi><mml:mo>:</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow><mml:mi>L</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2264;</mml:mo><mml:mi>K</mml:mi></mml:mrow></mml:munder><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<p>By training the neural network, a set of <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math>
</inline-formula> with parameter <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mi>w</mml:mi></mml:math>
</inline-formula> can be obtained. Due to the strong fitting ability of the neural network, such a set of <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math>
</inline-formula> can be highly approximated <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>p</mml:mi><mml:mo>&#x2225;</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:msub><mml:mo>&#x2225;</mml:mo><mml:mi>L</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2264;</mml:mo><mml:mi>K</mml:mi></mml:math>
</inline-formula>. At the same time, under the condition that <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mi>w</mml:mi></mml:math>
</inline-formula> does not exceed a certain range, formula (6) can be obtained:</p>
<p><disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">E</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<p>Formula (6) is the adversarial loss of the discriminator.</p>
<p>For the classification loss, the classification loss uses cross entropy because part of the convolutional neural network and the fully connected classifier at the end respectively complete independent classification tasks.</p>
<p>The discriminator is an N&#x002B;1-dimensional classifier, its input is data samples, and its output is an N&#x002B;1-dimensional vector: <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">c</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">k</mml:mi></mml:mrow><mml:mo>&#x002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
</inline-formula>. Logical vector can be expressed as class probability, formula (7) expresses the probability that <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mi>x</mml:mi></mml:math>
</inline-formula> is true and belongs to class <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mi>i</mml:mi></mml:math>
</inline-formula>. Therefore, the classification loss of the discriminator can be expressed as formula (8)</p>
<p><disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>d</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math>
</disp-formula></p>
<p><disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8"><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>d</mml:mi><mml:mi>e</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x2223;</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x003C;</mml:mo><mml:mi>N</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mo movablelimits="false">&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mrow><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">x</mml:mi><mml:mi mathvariant="normal">p</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>c</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math>
</disp-formula></p>
<p>b. Discriminator network structure</p>
<p>As shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, the discriminator D is composed of 3 convolutional layers and 2 fully connected layers in series. The real traffic sample PBV with a length of 1480 through data preprocessing is used to form and generate a three-dimensional tensor (20&#x002A;74&#x002A;1) that is consistent with the flow through dimensional transformation (reshape). This tensor is then sent to the 3-layer convolution kernel <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:mi>w</mml:mi></mml:math>
</inline-formula> with a size of 3&#x002A; 3 in the convolutional layer. LeakyReLU is used for activation after each convolution and can retain a small slope in the negative half axis; this slope is set equal to 0.2 in this article. Compared to the ReLU activation function, LeakyReLU can prevent the disappearance of the gradient during training. After flattening through the flatten layer, the tensor is input to the fully connected network, which is built using a stacked network with shared weights using Lambda to call a custom activation function. This network then produces predictions of authenticity using softmax to output normalized category probabilities.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>SACGAN discriminant network structure diagram</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-3.png"/>
</fig>
</sec>
<sec id="s3_2_3">
<label>3.2.3</label>
<title>Generator Structure</title>
<p>a. Loss function</p>
<p>To solve the defect of Jensen-Shannon (JS) divergence, the Wasserstein distance is used, which represents the EM distance from the real data set <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math>
</inline-formula> to the generated data set <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:math>
</inline-formula> [<xref ref-type="bibr" rid="ref-39">39</xref>]. The loss function of the generator is part of the adversarial loss function of the discriminator.</p>
<p><disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>G</mml:mi></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x223C;</mml:mo><mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi mathvariant="normal">g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>f</mml:mi><mml:mi>w</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<p>b. Generator network structure</p>
<p>Network parameters are shown in <xref ref-type="fig" rid="fig-4">Fig. 4</xref>. The generator G is constructed using a 1-layer deconvolution network and a 1-layer convolution network. First, 100-dimension random noise <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mi>z</mml:mi></mml:math>
</inline-formula> that conforms to the Gaussian distribution is input into the fully connected network. This noise is then converted into a three-dimensional tensor via dimensional transformation (reshaping). The dimensionally transformed tensor is then input into the convolution kernel <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:mi>w</mml:mi></mml:math>
</inline-formula> with a size of 4&#x002A;4 and a stride deconvolution layer of 2, which is activated by LeakyReLU and then input to a convolution layer <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mi>w</mml:mi></mml:math>
</inline-formula> with a kernel size of 7&#x002A;7 and stride of 2, using tanh activation to generate a tensor of (20&#x002A;74&#x002A;1).</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>SACGAN generated network structure diagram</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-4.png"/>
</fig>
</sec>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>SACGAN-Based Semisupervised Encryption Flow Identification Method</title>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Data Preprocessing</title>
<p>We capture original data packets in the PCAP (a file format, saved by Wireshark software) format and preprocess them as inputs for subsequent model training. Typically, PCAP preprocessing has four steps: filtering, truncation/zero padding, normalization and packet labeling.</p>
<p>a. Filtering:</p>
<p>First, we delete the ethernet header of the original data packet, data link layer information such as mac address, frame type, etc., and discard packets without application layer data. Filtering reduces the size of the input data packet. Additionally, to obtain better performance, noise is also filtered.</p>
<p>b. Truncation and zero padding:</p>
<p>To fix the input size of each data packet in the model, the TCP header is truncated to 20 bytes, and the UDP header is zero-filled to 20 bytes. The application layer data is truncated and zero-filled to form a fixed-length data packet. Because most data packets are less than or equal to 1460 bytes in length, the data length of the application layer is set to 1460 bytes. Finally, data with a length of 1480 bytes is formed.</p>
<p>c. Normalization</p>
<p>Each processed data is sent as a Packet Byte Vector (PBV). For example, <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mrow><mml:msub><mml:mi>i</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math>
</inline-formula> PBV is described as follows:</p>
<p><disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
</disp-formula></p>
<p>where <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math>
</inline-formula> represents the <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mrow><mml:msub><mml:mi>j</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math>
</inline-formula> <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:mrow><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mspace width="thickmathspace"></mml:mspace></mml:mrow></mml:math>
</inline-formula>byte. To converge faster, each <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mi>P</mml:mi></mml:math>
</inline-formula> is normalized to [0,1].</p>
<p>d. Packet labeling</p>
<p>Finally, the traffic that can be marked in each packet is labeled with the corresponding application to form a labeled sample, and the remainders are merged into unlabeled samples to form a data set for model training.</p>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Model Training</title>
<p>SACGAN training uses a combination of supervised and unsupervised losses, which can improve model learning. Because the adversarial loss <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mspace width="thickmathspace"></mml:mspace></mml:mrow></mml:math>
</inline-formula> and classification loss <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow></mml:math>
</inline-formula> are often not an order of magnitude in value, and because the total network has independent branches, it is not reasonable to update the parameters of the entire network at one time. Therefore, the following parameter rules are used when training the model:<list list-type="bullet"><list-item>
<p>When optimizing <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:math>
</inline-formula>, the network parameters G, CNN and MLP_S are updated;</p></list-item><list-item>
<p>When optimizing <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow></mml:math>
</inline-formula>, the network parameters G, CNN and MLP_C are updated;</p></list-item><list-item>
<p>During one update, the generator is updated twice, the feature extraction network is updated twice, and the branch network is updated once.</p></list-item></list></p>
<p>Using the gradient penalty to combat loss, the loss <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:math>
</inline-formula> does not go through the sigmoid function to avoid the unreasonable distance between the generated distribution and the real distribution caused by JS divergence. Simultaneously, it solves the problem that the ACGAN network updates the classification loss and the adversarial loss at the same time when updating the discriminator network parameters.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiments</title>
<sec id="s4_1">
<label>4.1</label>
<title>Experimental Setup</title>
<p>The experiment in this chapter uses the ISCX VPN-nonVPN and USTC-TFC2016 data sets, as shown in <xref ref-type="table" rid="table-1">Tab. 1</xref>. We select 10,000 pieces of data randomly from 9 applications of USTC-TCF2016.</p>
<table-wrap id="table-1"><label>Table 1</label>
<caption>
<title>Data set description</title></caption>
<table><colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th colspan="3">ISCX VPN-nonVPN</th>
<th colspan="3">USTC-TFC2016</th>
</tr>
</thead>
<tbody>
<tr>
<td>Application Name</td>
<td>Quantity</td>
<td>Proportion (%)</td>
<td>Application Name</td>
<td>Quantity</td>
<td>Proportion (%)</td>
</tr>
<tr>
<td>AIM_chat</td>
<td>4869</td>
<td>2.356</td>
<td>BitTorrent</td>
<td>10000</td>
<td>0.11</td>
</tr>
<tr>
<td>Email</td>
<td>4417</td>
<td>2.137</td>
<td>FTP</td>
<td>10000</td>
<td>0.11</td>
</tr>
<tr>
<td>Facebook</td>
<td>5527</td>
<td>2.674</td>
<td>Gmail</td>
<td>10000</td>
<td>0.11</td>
</tr>
<tr>
<td>Gmail</td>
<td>7329</td>
<td>3.546</td>
<td>Mysql</td>
<td>10000</td>
<td>0.11</td>
</tr>
<tr>
<td>Hangout</td>
<td>7587</td>
<td>3.671</td>
<td>Outlook</td>
<td>10000</td>
<td>0.11</td>
</tr>
<tr>
<td>ICQ</td>
<td>4243</td>
<td>2.053</td>
<td>SMB</td>
<td>10000</td>
<td>0.11</td>
</tr>
<tr>
<td>Netflix</td>
<td>51932</td>
<td>25.126</td>
<td>Skype</td>
<td>10000</td>
<td>0.11</td>
</tr>
<tr>
<td>SCPdown</td>
<td>15390</td>
<td>7.446</td>
<td>WOW</td>
<td>10000</td>
<td>0.11</td>
</tr>
<tr>
<td>SFTPDown</td>
<td>4729</td>
<td>2.287</td>
<td>Weibo</td>
<td>10000</td>
<td>0.11</td>
</tr>
<tr>
<td>Skype</td>
<td>4607</td>
<td>2.229</td>
<td></td>
<td></td>
<td></td>
</tr>
<tr>
<td>Spotify</td>
<td>14442</td>
<td>6.987</td>
<td></td>
<td></td>
<td></td>
</tr>
<tr>
<td>TorTwitter</td>
<td>14654</td>
<td>7.089</td>
<td></td>
<td></td>
<td></td>
</tr>
<tr>
<td>Vimeo</td>
<td>18755</td>
<td>9.074</td>
<td></td>
<td></td>
<td></td>
</tr>
<tr>
<td>Voipbuster</td>
<td>35469</td>
<td>17.161</td>
<td></td>
<td></td>
<td></td>
</tr>
<tr>
<td>Youtube</td>
<td>12738</td>
<td>6.163</td>
<td></td>
<td></td>
<td></td>
</tr>
<tr>
<td>Total</td>
<td>206688</td>
<td>100</td>
<td>Total</td>
<td>90000</td>
<td>100</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>For comparison, two data sets are investigated experimentally using the CNN-based classification model [<xref ref-type="bibr" rid="ref-14">14</xref>] and the SACGAN-based classification model. Before experimentation, we selected a specified number of labeled data from the data set through the code and used the remainder of the data set as unlabeled data for the experiment.</p>
<p>The network structure of the SACGAN&#x2019;s discriminator and generator is shown in Section 3.2. On the selected data set, the batch gradient descent method is used for training, with a batch size of 256. Noise is randomly sampled from the uniform distribution of [&#x2212;1, 1] with a sample size of 100. The learning rate parameter is 0.0002, and the Adam optimizer is used to optimize the loss function of the generator and the discriminator.</p>
<p>The CNN network structure used is shown in reference Wang et al. [<xref ref-type="bibr" rid="ref-14">14</xref>]. The batch gradient descent method is used to train the selected data set and the batch size 128. The Rmsprop optimizer is used to optimize the cross entropy loss function.</p>
<p>This article uses a SACGAN to perform classification experiments when there are only 1000, 2000, 3000 and 4000 labeled data in each class sample. Under the same conditions, this network is compared to the classification results of CNN to demonstrate the superior performance of the SACGAN in semisupervised encryption traffic classification.In the experiment, the data set was split into a training set and test set according to the ratio of 6:4 for cross-validation. The results in section 4.3 are based on the test set.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Evaluation Index</title>
<p>To evaluate the performance of the model, we use the following two types of indicators:</p>
<p>a. Precision and recall</p>
<p>False positive (FP) indicates that the traffic of noncategory C, where C refers to a specific category, is classified as category C. True Negative (TN) indicates that the traffic of noncategory C is classified as noncategory C. False negative (FN) indicates that traffic belonging to category C is classified as noncategory C. True positive (TP) indicates that traffic belonging to category C is classified as category C.</p>
<p>The precision rate (henceforth &#x201C;precision&#x201D;) is calculated using formula (11), and the recall rate (henceforth &#x201C;recall&#x201D;) is calculated using formula (13):</p>
<p><disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:mrow><mml:mi mathvariant="normal">P</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mspace width="thickmathspace"></mml:mspace></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
</disp-formula></p>
<p><disp-formula id="eqn-12"><label>(12)</label><mml:math id="mml-eqn-12" display="block"><mml:mrow><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
</disp-formula></p>
<p>b. F1-Score</p>
<p>The F1score is a weighted and average of the precision and recall rates, and is used to comprehensively describe the entire index [<xref ref-type="bibr" rid="ref-42">42</xref>]. The most common formula for calculating the F1-scorescore is:</p>
<p><disp-formula id="eqn-13"><label>(13)</label><mml:math id="mml-eqn-13" display="block"><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mi mathvariant="normal">S</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mspace width="thickmathspace"></mml:mspace></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mspace width="thickmathspace"></mml:mspace><mml:mi mathvariant="normal">P</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mspace width="thickmathspace"></mml:mspace></mml:mrow><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mspace width="thickmathspace"></mml:mspace><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mspace width="thickmathspace"></mml:mspace></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mspace width="thickmathspace"></mml:mspace><mml:mi mathvariant="normal">P</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mspace width="thickmathspace"></mml:mspace></mml:mrow><mml:mo>&#x002B;</mml:mo><mml:mrow><mml:mspace width="thickmathspace"></mml:mspace><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mspace width="thickmathspace"></mml:mspace></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
</disp-formula></p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Experimental Results</title>
<sec id="s4_3_1">
<label>4.3.1</label>
<title>Performance Result</title>
<p><xref ref-type="fig" rid="fig-5">Fig. 5</xref> shows the change in the confrontation loss on the SACGAN discriminator and generator. The confrontation loss of the discriminator gradually decreases, the confrontation loss of the generator gradually increases, and both stabilize after approximately 2000 rounds.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>SACGAN training loss (a) Training loss of &#x201C;ISCX VPN-nonVPN&#x201D; dataset (b) Training loss of &#x201C;USTC-TFC2016&#x201D; dataset</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-5.png"/>
</fig>
</sec>
<sec id="s4_3_2">
<label>4.3.2</label>
<title>Classification Result</title>
<p>As shown in <xref ref-type="table" rid="table-2">Tab. 2</xref>, we use the SACGAN and CNN to perform classification experiments with 1000, 2000, 3000, and 4000 labeled samples, the following classification results are produced.</p>
<p>When the number of labeled samples is 1000, the classification accuracy of SACGAN is improved by approximately 5% compared to the CNN. When the number of labeled samples is 2000, it is improved by approximately 3% compared to the CNN. When the number of labeled samples is 3000, it is improved by below 1% compared to the CNN. When the number of labeled samples is 4000, the classification accuracy of the SACGAN is similar to that of the CNN.</p>
<p>Accuracy from the SACGAN remains poor regarding classification performance. We use the evaluation indicators introduced in Section 4.2 to analyze the improvement of each application more comprehensively.</p>
<table-wrap id="table-2"><label>Table 2</label>
<caption>
<title>Classification accuracy</title></caption>
<table><colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th rowspan="2">Number of labeled samples<break/>(per application)</th>
<th colspan="2">ISCX VPN-nonVPN</th>
<th colspan="2">USTC-TFC2016</th>
</tr>
<tr>
<th>SACGAN</th>
<th>CNN</th>
<th>SACGAN</th>
<th>CNN</th>
</tr>
</thead>
<tbody>
<tr>
<td>1000</td>
<td>92.15%</td>
<td>88.25%</td>
<td>99.30%</td>
<td>95.5%</td>
</tr>
<tr>
<td>2000</td>
<td>92.92%</td>
<td>89.60%</td>
<td>99.41%</td>
<td>97.9%</td>
</tr>
<tr>
<td>3000</td>
<td>93.10%</td>
<td>92.40%</td>
<td>99.53%</td>
<td>98.9%</td>
</tr>
<tr>
<td>4000</td>
<td>93.18%</td>
<td>93.30%</td>
<td>99.58%</td>
<td>99.35%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>a. Precision:</p>
<p>As shown in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>, in addition to the two applications of Facebook and Hangout, SACGAN&#x2019;s precision index is basically better than that of CNN when there are labeled samples in 1000, 2000, and 3000. Among them, the improvement of AIM_Chat is greater, and the improvement is close to 20%; ICQ has improved significantly, with an increase of close to 5%; other applications such as Netflix, SCPDown, Skype, Spotify, Tort Witter, VoIP buster, and YouTube also have a small increase.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>ISCX data set precision comparison (a) 1000 labeled data (b) 2000 labeled data (c) 3000 labeled data (d) 4000 labeled data</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-6.png"/>
</fig>
<p>As shown in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>, SACGAN&#x2019;s precision index is basically better than CNN when the number of labeled samples is 1000, 2000, 3000. Among them, Gmail increases by about 10% when there are 1000 labeled samples, and about 5% when there are 2000 labeled samples. BitTorrent, Outlook, and Skype improve significantly, with an increase of about 5% when there are 1000 labeled samples, and an increase of about 2% when the number of labeled samples is 2000 and 3000. Other applications also have a small improvement when there are 1000&#x007E;3000 labeled samples.</p>
<p>b. Recall:</p>
<p>As shown in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>, in the experiments of 1000, 2000 and 3000 labeled samples, the recall of Gmail, netfliex, SCPDown, SFTPDown, Skype, Spotify, TorTwitter, Vimeo, VoipBuster, and YouTube are better than CNN. Among them, the improvement of Gmail is more significant, close to 20%.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>USTC data set precision comparison (a) 1000 labeled data (b) 2000 labeled data (c) 3000 labeled data (d) 4000 labeled data</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-7.png"/>
</fig>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>ISCX data set recall comparison (a) 1000 labeled data (b) 2000 labeled data (c) 3000 labeled data (d) 4000 labeled data</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-8.png"/>
</fig>
<p>As shown in <xref ref-type="fig" rid="fig-9">Fig. 9</xref>, with 1000, 2000, and 3000 labeled samples, except for BitTorrent, SACGAN&#x2019;s recall index is better than CNN. Among them, the improvement of Gmail is significant, with an increase of about 5%. With 1000 and 2000 labeled samples, Outlook has improved significantly, with an increase of about 2.5%.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>USTC data set recall comparison (a) 1000 labeled data (b) 2000 labeled data (c) 3000 labeled data (d) 4000 labeled data</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-9.png"/>
</fig>
<fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>ISCX data set recall comparison (a) 1000 labeled data (b) 2000 labeled data (c) 3000 labeled data (d) 4000 labeled data</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-10.png"/>
</fig>
<p>c. F1-score:</p>
<p>As shown in <xref ref-type="fig" rid="fig-10">Fig. 10</xref>, similar to the precision index, except for Facebook and Hangout, SACGAN&#x2019;s f1-score index is basically better than CNN when the number of labeled samples is 1000, 2000, and 3000. Among them, AIM_Chat has a greater improvement, which is about 5% to 10%. Email, Gmail, and ICQ have been significantly improved, and the increase is large. Other applications such as Netflix, SCPDown, Skype, Spotify, TorTwitter, VoipBuster, YouTube also have a small increase.</p>
<p>As shown in <xref ref-type="fig" rid="fig-11">Fig. 11</xref> similar to the precision index, when the number of labeled samples of SACGAN is 1000, 2000, and 3000, the f1-score index is basically better than that of CNN. Among them, the improvement of Gmail is greater, which is about 3%-12%; Skype has a significant improvement, an increase of about 1%&#x007E;5%; other applications also have a small increase.</p>
<p>From the above experimental results, it can be seen that when there is less labeled data, the classification accuracy of most applications of SACGAN is greatly improved compared to CNN.</p>
<fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>USTC data set recall comparison (a) 1000 labeled data (b) 2000 labeled data (c) 3000 labeled data (d) 4000 labeled data</title></caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CSSE_18086-fig-11.png"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusions</title>
<p>In this paper, we modified the network structure of the auxiliary classification generation adversarial network and modified the loss function of the generator and the discriminator. We designed a semi-supervised encrypted traffic recognition scheme based on the auxiliary classification Generative Adversarial Network. Using the proposed method and the classification method of CNN in the paper [<xref ref-type="bibr" rid="ref-14">14</xref>], classification experiments were carried out on two data sets: ISCX and USTC. The results show that under the same conditions, the proposed method has higher recognition accuracy when there are fewer labeled samples.</p>
</sec>
</body>
<back>
<ack>
<p>We thank the anonymous reviewers and editors for their very constructive comments.</p>
</ack><fn-group>
<fn fn-type="other">
<p><bold>Funding Statement:</bold> This work is supported by the Science and Technology Project of State Grid Jiangsu Electric Power Co., Ltd. under Grant No. J2020068.</p>
</fn>
<fn fn-type="conflict">
<p><bold>Conflicts of Interest:</bold> The authors declare that they have no conflicts of interest to report regarding the present study.</p>
</fn>
</fn-group>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K. C.</given-names> <surname>Claffy</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Braun</surname></string-name> and <string-name><given-names>G. C.</given-names> <surname>Polyzos</surname></string-name></person-group>, &#x201C;<article-title>A parameterizable methodology for Internet traffic flow profiling</article-title>,&#x201D; <source>IEEE Journal on Selected Areas in Communications</source>, vol. <volume>13</volume>, no. <issue>8</issue>, pp. <fpage>1481</fpage>&#x2013;<lpage>1494</lpage>, <year>1995</year>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Gu</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Sun</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Ji</surname></string-name></person-group>, &#x201C;<article-title>Fast traffic classification using joint distribution of packet size and estimated protocol processing time</article-title>,&#x201D; <source>IEICE Transactions on Information and System</source>, vol. <volume>93</volume>, no. <issue>11</issue>, pp. <fpage>2944</fpage>&#x2013;<lpage>2952</lpage>, <year>2010</year>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>S. H.</given-names> <surname>Yeganeh</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Eftekhar</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Ganjali</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Keralapura</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Nucci</surname></string-name></person-group>, &#x201C;<article-title>CUTE: Traffic classification using terms</article-title>,&#x201D; in <conf-name>Proc. ICCCN</conf-name>, <publisher-loc>Munich, MUC, Germany</publisher-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>9</lpage>, <year>2012</year>. </mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H. A. H.</given-names> <surname>Ibrahim</surname></string-name>, <string-name><given-names>S. M.</given-names> <surname>Nor</surname></string-name> and <string-name><given-names>H. A.</given-names> <surname>Jamil</surname></string-name></person-group>, &#x201C;<article-title>Online hybrid internet traffic classification algorithm based on signature statistical and port methods to identify internet applications</article-title>,&#x201D; in <conf-name>Proc. ICCSCE</conf-name>, <publisher-loc>Penang, PEN, Malaysia</publisher-loc>, pp. <fpage>185</fpage>&#x2013;<lpage>190</lpage>, <year>2013</year>. </mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Conti</surname></string-name>, <string-name><given-names>L. V.</given-names> <surname>Mancini</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Spolaor</surname></string-name> and <string-name><given-names>N. V.</given-names> <surname>Verde</surname></string-name></person-group>, &#x201C;<article-title>Can&#x2019;t you hear me knocking: Identification of user actions on android apps via traffic analysis</article-title>,&#x201D; in <conf-name>Proc. CODASPY</conf-name>, <publisher-loc>New York, NY, USA</publisher-loc>, pp. <fpage>297</fpage>&#x2013;<lpage>304</lpage>, <year>2019</year>. </mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Kim</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Shin</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Han</surname></string-name></person-group>, &#x201C;<article-title>Analysis of feature importance and interpretation for malware classification</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>65</volume>, no. <issue>3</issue>, pp. <fpage>1891</fpage>&#x2013;<lpage>1904</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Du</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Si</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Guo</surname></string-name> and <string-name><given-names>T.</given-names> <surname>Jin</surname></string-name></person-group>, &#x201C;<article-title>Using object detection network for malware detection and identification in network traffic packets</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>64</volume>, no. <issue>3</issue>, pp. <fpage>1785</fpage>&#x2013;<lpage>1796</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Mo</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Xiaojuan</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Mingshu</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Lei</surname></string-name> and <string-name><given-names>K.</given-names> <surname>Javeed</surname></string-name></person-group>, &#x201C;<article-title>A network traffic classification model based on metric learning</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>64</volume>, no. <issue>2</issue>, pp. <fpage>941</fpage>&#x2013;<lpage>959</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Sheng</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Zeng</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Ye</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>HAST-IDS: Learning hierarchical spatial-temporal features using deep neural networks to improve intrusion detection</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>6</volume>, pp. <fpage>1792</fpage>&#x2013;<lpage>1806</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Aceto</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Ciuonzo</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Montieri</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Pescap&#x00E9;</surname></string-name></person-group>, &#x201C;<article-title>Mobile encrypted traffic classification using deep learning</article-title>,&#x201D; in <conf-name>Proc. TMA</conf-name>, <publisher-loc>Vienna, VIE, AT</publisher-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>8</lpage>, <year>2018</year>. </mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Aceto</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Ciuonzo</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Montieri</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Pescap&#x00E9;</surname></string-name></person-group>, &#x201C;<article-title>Mobile encrypted traffic classification using deep learning: Experimental evaluation, lessons learned and challenges</article-title>,&#x201D; <source>IEEE Transactions on Network and Service Management</source>, vol. <volume>16</volume>, no. <issue>2</issue>, pp. <fpage>445</fpage>&#x2013;<lpage>458</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Lopez-Martin</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Carro</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Sanchez-Esguevillas</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Lloret</surname></string-name></person-group>, &#x201C;<article-title>Network traffic classifier with convolutional and recurrent neural networks for internet of things</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>5</volume>, pp. <fpage>18042</fpage>&#x2013;<lpage>18050</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>H&#x00F6;chst</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Baumg&#x00E4;rtner</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Hollick</surname></string-name> and <string-name><given-names>B.</given-names> <surname>Freisleben</surname></string-name></person-group>, &#x201C;<article-title>Unsupervised traffic flow classification using a neural autoencoder</article-title>,&#x201D; in <conf-name>Proc. LCN</conf-name>, <publisher-loc>Singapore, SGP</publisher-loc>, pp. <fpage>523</fpage>&#x2013;<lpage>526</lpage>, <year>2017</year>. </mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>P.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Ye</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Chen</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Qian</surname></string-name></person-group>, &#x201C;<article-title>Datanet: Deep learning based encrypted network traffic classification in SDN home gateway</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>6</volume>, pp. <fpage>55380</fpage>&#x2013;<lpage>55391</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>I. J.</given-names> <surname>Goodfellow</surname></string-name></person-group>, &#x201C;<article-title>Generative adversarial networks</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1406.2661</comment>, <year>2014</year>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Ledig</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Theis</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Huszar</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Caballero</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Cunningham</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Photo-realistic single image super-resolution using a generative adversarial network</article-title>,&#x201D; in <conf-name>Proc. CVPR</conf-name>, <publisher-loc>Honolulu, HNL, USA</publisher-loc>, pp. <fpage>105</fpage>&#x2013;<lpage>114</lpage>, <year>2017</year>. </mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H. W.</given-names> <surname>Dong</surname></string-name>, <string-name><given-names>W. Y.</given-names> <surname>Hsiao</surname></string-name>, <string-name><given-names>L. C.</given-names> <surname>Yang</surname></string-name> and <string-name><given-names>Y. H.</given-names> <surname>Yang</surname></string-name></person-group>, &#x201C;<article-title>MuseGAN: Multi-track sequential generative adversarial networks for symbolic music generation and accompaniment</article-title>,&#x201D; in <conf-name>Proc. AAAI</conf-name>, <publisher-loc>New Orleans, NO, USA</publisher-loc>, pp. <fpage>117</fpage>&#x2013;<lpage>125</lpage>, <year>2018</year>. </mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>O.</given-names> <surname>Olabiyi</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Salimov</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Khazane</surname></string-name> and <string-name><given-names>E. T.</given-names> <surname>Mueller</surname></string-name></person-group>, &#x201C;<article-title>Multi-turn dialogue response generation in an adversarial learning framework</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1805.11752</comment>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Zhao</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Ying</surname></string-name>, <string-name><given-names>Q.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Noel</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Human face sketch to rgb image with edge optimization and generative adversarial networks</article-title>,&#x201D; <source>Intelligent Automation &#x0026; Soft Computing</source>, vol. <volume>26</volume>, no. <issue>4</issue>, pp. <fpage>1391</fpage>&#x2013;<lpage>1401</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Chen</surname></string-name> and <string-name><given-names>Z.</given-names> <surname>Sha</surname></string-name></person-group>, &#x201C;<article-title>Edge detection based on generative adversarial networks</article-title>,&#x201D; <source>Journal of New Media</source>, vol. <volume>2</volume>, no. <issue>2</issue>, pp. <fpage>61</fpage>&#x2013;<lpage>77</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Hu</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Tan</surname></string-name></person-group>, &#x201C;<article-title>Generating adversarial malware examples for black-box attacks based on GAN</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1702.05983</comment>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>J. Y.</given-names> <surname>Kim</surname></string-name>, <string-name><given-names>S. J.</given-names> <surname>Bu</surname></string-name> and <string-name><given-names>S. B.</given-names> <surname>Cho</surname></string-name></person-group>, &#x201C;<article-title>Malware detection using deep transferred generative adversarial networks</article-title>,&#x201D; in <conf-name>Proc. ICONIP</conf-name>, <publisher-loc>Guangzhou, GZ, China</publisher-loc>, pp. <fpage>556</fpage>&#x2013;<lpage>564</lpage>, <year>2017</year>. </mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>Z.</given-names> <surname>Lin</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Shi</surname></string-name> and <string-name><given-names>Z.</given-names> <surname>Xue</surname></string-name></person-group>, &#x201C;<article-title>IDSGAN: Generative adversarial networks for attack generation against intrusion detection</article-title>,&#x201D; <comment>arXiv e-print, arXiv:1809.02077</comment>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Vu</surname></string-name>, <string-name><given-names>C. Thanh</given-names> <surname>Bui</surname></string-name> and <string-name><given-names>U.</given-names> <surname>Nguyen</surname></string-name></person-group>, &#x201C;<article-title>A deep learning based method for handling imbalanced problem in network traffic classification</article-title>,&#x201D; in <conf-name>Proc. SoICT</conf-name>, <publisher-loc>Nha Trang City, NTC</publisher-loc>: <publisher-name>Viet Nam</publisher-name>, pp. <fpage>333</fpage>&#x2013;<lpage>339</lpage>, <year>2017</year>. </mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Douzas</surname></string-name> and <string-name><given-names>F.</given-names> <surname>Bacao</surname></string-name></person-group>, &#x201C;<article-title>Effective data generation for imbalanced learning using conditional generative adversarial networks</article-title>,&#x201D; <source>Expert Systems with Applications</source>, vol. <volume>91</volume>, no. <issue>2</issue>, pp. <fpage>464</fpage>&#x2013;<lpage>471</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Arjovsky</surname></string-name> and <string-name><given-names>L.</given-names> <surname>Bottou</surname></string-name></person-group>, &#x201C;<article-title>Towards principled methods for training generative adversarial networks</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1701.04862</comment>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Arjovsky</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Chintala</surname></string-name> and <string-name><given-names>L.</given-names> <surname>Bottou</surname></string-name></person-group>, &#x201C;<article-title>Wasserstein generative adversarial networks</article-title>,&#x201D; in <conf-name>Proc. ICML</conf-name>, <publisher-loc>Sydney, SYD, AUS</publisher-loc>, pp. <fpage>214</fpage>&#x2013;<lpage>223</lpage>, <year>2017</year>. </mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>U.</given-names> <surname>Fiore</surname></string-name>, <string-name><given-names>A. De</given-names> <surname>Santis</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Perla</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Zanetti</surname></string-name> and <string-name><given-names>F.</given-names> <surname>Palmieri</surname></string-name></person-group>, &#x201C;<article-title>Using generative adversarial networks for improving classification effectiveness in credit card fraud detection</article-title>,&#x201D; <source>Information Sciences</source>, vol. <volume>479</volume>, no. <issue>4</issue>, pp. <fpage>448</fpage>&#x2013;<lpage>455</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Zheng</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Zhu</surname></string-name>, <string-name><given-names>Y. H.</given-names> <surname>Tang</surname></string-name>, <string-name><given-names>M. J.</given-names> <surname>Tang</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Conditional wasserstein generative adversarial network-gradient penalty-based approach to alleviating imbalanced data classification</article-title>,&#x201D; <source>Information Sciences</source>, vol. <volume>512</volume>, pp. <fpage>1009</fpage>&#x2013;<lpage>1023</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Rasmus</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Berglund</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Honkala</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Valpola</surname></string-name> and <string-name><given-names>T.</given-names> <surname>Raiko</surname></string-name></person-group>, &#x201C;<chapter-title>Semi-supervised learning with ladder networks</chapter-title>,&#x201D; in <source>Advances in neural information processing systems</source>, pp. <fpage>3546</fpage>&#x2013;<lpage>3554</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Johnson</surname></string-name> and <string-name><given-names>T.</given-names> <surname>Zhang</surname></string-name></person-group>, &#x201C;<article-title>Supervised and semi-supervised text categorization using LSTM for region embeddings</article-title>,&#x201D; in <conf-name>Proc. ICML</conf-name>, <publisher-loc>New York, NY, USA</publisher-loc>, pp. <fpage>526</fpage>&#x2013;<lpage>534</lpage>, <year>2016</year>. </mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Rezaei</surname></string-name> and <string-name><given-names>X.</given-names> <surname>Liu</surname></string-name></person-group>, &#x201C;<article-title>How to achieve high classification accuracy with Just a few labels: A semi-supervised approach using sampled packets</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1812.09761</comment>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Salimans</surname></string-name>, <string-name><given-names>I.</given-names> <surname>Goodfellow</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Zaremba</surname></string-name>, <string-name><given-names>V.</given-names> <surname>Cheung</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Radford</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Improved techniques for training GANs</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1606.03498</comment>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Z.</given-names> <surname>Dai</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Yang</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Yang</surname></string-name> and <string-name><given-names>W. W.</given-names> <surname>Cohen</surname></string-name></person-group>, &#x201C;<article-title>Good semi-supervised learning that requires a bad GAN</article-title>,&#x201D; in <conf-name>Proc. NIPS</conf-name>, <publisher-loc>New York, NY, USA</publisher-loc>, pp. <fpage>6513</fpage>&#x2013;<lpage>6523</lpage>, <year>2017</year>. </mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>J. T.</given-names> <surname>Springenberg</surname></string-name></person-group>, &#x201C;<article-title>Unsupervised and semi-supervised learning with categorical generative adversarial networks</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1511.06390</comment>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A. S.</given-names> <surname>Iliyasu</surname></string-name> and <string-name><given-names>H.</given-names> <surname>Deng</surname></string-name></person-group>, &#x201C;<article-title>Semi-supervised encrypted traffic classification with deep convolutional generative adversarial networks</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>8</volume>, pp. <fpage>118</fpage>&#x2013;<lpage>126</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Odena</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Olah</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Shlens</surname></string-name></person-group>, &#x201C;<article-title>Conditional image synthesis with auxiliary classifier GANs</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1610.09585</comment>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Radford</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Metz</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Chintala</surname></string-name></person-group>, &#x201C;<article-title>Unsupervised representation learning with deep convolutional generative adversarial networks</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1511.06434</comment>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Arjovsky</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Chintala</surname></string-name> and <string-name><given-names>L.</given-names> <surname>Bottou</surname></string-name></person-group>, &#x201C;<article-title>Wasserstein GAN</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1701.07875</comment>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Salimans</surname></string-name>, <string-name><given-names>I.</given-names> <surname>Goodfellow</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Zaremba</surname></string-name>, <string-name><given-names>V.</given-names> <surname>Cheung</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Radford</surname></string-name></person-group>, &#x201C;<article-title>Improved techniques for training GANs</article-title>,&#x201D; <comment>arXiv e-prints, arXiv:1606.03498</comment>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Xu</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Chen</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Li</surname></string-name></person-group>, &#x201C;<article-title>Empirical evaluation of rectified activations in convolutional network</article-title>,&#x201D; <comment>arXiv e-print, arXiv:1505.00853</comment>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Kim</surname></string-name>, <string-name><given-names>K. C.</given-names> <surname>Claffy</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Fomenkov</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Barman</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Faloutsos</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Internet traffic classification demystified: Myths, caveats and the best practices</article-title>,&#x201D; in <conf-name>Proc. CoNEXT</conf-name>, <publisher-loc>Madrid, Spain, </publisher-loc><year>2008</year>. </mixed-citation></ref>
</ref-list>
</back>
</article>