<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">28743</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2023.028743</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Two-Stream Deep Learning Architecture-Based Human Action Recognition</article-title>
<alt-title alt-title-type="left-running-head">Two-Stream Deep Learning Architecture-Based Human Action Recognition</alt-title>
<alt-title alt-title-type="right-running-head">Two-Stream Deep Learning Architecture-Based Human Action Recognition</alt-title>
</title-group>
<contrib-group content-type="authors">
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Shehzad</surname><given-names>Faheem</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Khan</surname><given-names>Muhammad Attique</given-names>
</name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Yar</surname><given-names>Muhammad Asfand E.</given-names>
</name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Sharif</surname><given-names>Muhammad</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Alhaisoni</surname><given-names>Majed</given-names>
</name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Tariq</surname><given-names>Usman</given-names>
</name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Majumdar</surname><given-names>Arnab</given-names>
</name><xref ref-type="aff" rid="aff-6">6</xref></contrib>
<contrib id="author-8" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Thinnukool</surname><given-names>Orawit</given-names>
</name><xref ref-type="aff" rid="aff-7">7</xref><email>orawit.t@cmu.ac.th</email></contrib>
<aff id="aff-1"><label>1</label><institution>Department of Computer Science, COMSATS University Islamabad</institution>, <addr-line>Wah Campus</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-2"><label>2</label><institution>Department of Computer Science, HITEC University</institution>, <addr-line>Taxila</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-3"><label>3</label><institution>Department of Computer Science, Bahria University</institution>, <addr-line>Islamabad</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-4"><label>4</label><institution>Computer Sciences Department, College of Computer and Information Sciences, Princess Nourah bint Abdulrahman University</institution>, <addr-line>Riyadh, 11671</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-5"><label>5</label><institution>College of Computer Engineering and Science, Prince Sattam Bin Abdulaziz University</institution>, <addr-line>Al-Kharaj, 11942</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-6"><label>6</label><institution>Faculty of Engineering, Imperial College London</institution>, <addr-line>London, SW7 2AZ</addr-line>, <country>UK</country></aff>
<aff id="aff-7"><label>7</label><institution>College of Arts</institution>, <addr-line>Media</addr-line>, <institution>and Technology, Chiang Mai University</institution>, <addr-line>Chiang Mai, 50200</addr-line>, <country>Thailand</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Orawit Thinnukool. Email: <email>orawit.t@cmu.ac.th</email></corresp>
</author-notes>
<pub-date publication-format="print" date-type="pub" iso-8601-date="2022-12-15"><day>15</day>
<month>12</month>
<year>2022</year></pub-date>
<volume>74</volume>
<issue>3</issue>
<fpage>5931</fpage>
<lpage>5949</lpage>
<history>
<date date-type="received">
<day>16</day>
<month>2</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>06</day>
<month>5</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2023 Shehzad et al.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Shehzad et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_28743.pdf"></self-uri>
<abstract>
<p>Human action recognition (HAR) based on Artificial intelligence reasoning is the most important research area in computer vision. Big breakthroughs in this field have been observed in the last few years; additionally, the interest in research in this field is evolving, such as understanding of actions and scenes, studying human joints, and human posture recognition. Many HAR techniques are introduced in the literature. Nonetheless, the challenge of redundant and irrelevant features reduces recognition accuracy. They also faced a few other challenges, such as differing perspectives, environmental conditions, and temporal variations, among others. In this work, a deep learning and improved whale optimization algorithm based framework is proposed for HAR. The proposed framework consists of a few core stages i.e., frames initial preprocessing, fine-tuned pre-trained deep learning models through transfer learning (TL), features fusion using modified serial based approach, and improved whale optimization based best features selection for final classification. Two pre-trained deep learning models such as InceptionV3 and Resnet101 are fine-tuned and TL is employed to train on action recognition datasets. The fusion process increases the length of feature vectors; therefore, improved whale optimization algorithm is proposed and selects the best features. The best selected features are finally classified using machine learning (ML) classifiers. Four publicly accessible datasets such as Ut-interaction, Hollywood, <bold>Free Viewpoint Action Recognition using Motion History Volumes (IXMAS)</bold>, and centre of computer vision (UCF) Sports, are employed and achieved the testing accuracy of 100%, 99.9%, 99.1%, and 100% respectively. Comparison with <bold>state of the art techniques</bold> (SOTA), the proposed method showed the improved accuracy.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Human action recognition</kwd>
<kwd>deep learning</kwd>
<kwd>transfer learning</kwd>
<kwd>fusion of multiple features</kwd>
<kwd>features optimization</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Human Action Recognition (HAR) is a critical research topic in machine learning and computer vision applications [<xref ref-type="bibr" rid="ref-1">1</xref>]. Because of its covariant properties, HAR has gained a lot of popularity in recent decades. Real-world applications of HAR include robotics, location estimation, sports analysis, pedestrian detection, human-computer interaction, video games, and video surveillance [<xref ref-type="bibr" rid="ref-2">2</xref>]. Several human actions such as pointing, running, pushing, boxing, kicking, hand waving, jogging, clapping, diving, and named a few more are recognized in the video sequences (a few samples shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>). These actions are recognized through computerized systems automatically and effectively [<xref ref-type="bibr" rid="ref-3">3</xref>]. Nowadays, many HAR methods are used like wireless network-based method, video-based method, and sensor-based method [<xref ref-type="bibr" rid="ref-4">4</xref>]. However, video-based HAR approaches are gaining popularity due to their high recognition rate and ease of usage. Furthermore, HAR is broadly used in different industrial applications [<xref ref-type="bibr" rid="ref-5">5</xref>]. During the past few years, the big breakthroughs have been witnessed in this field. Also, the research interest in this field is evolving like understanding of actions and scenes, studying the human joints, and human posture recognition. The precision of HAR has been raised due to the growing of learning-based <bold>artificial intelligence</bold> (AI) [<xref ref-type="bibr" rid="ref-6">6</xref>]. Even though various innovations are being witnessed in AI technology, but still there exist quite a few challenges also in this field. For their learning-based algorithms, this field requires large datasets and corresponding labels. Several videos are obtained from the YouTube platform and manually fine-tuned in terms of detail, actions, and comprehension. The manual recognition and understanding process is time-consuming and labor-intensive [<xref ref-type="bibr" rid="ref-7">7</xref>].</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Sample human actions frames collected from UT-Interaction dataset [<xref ref-type="bibr" rid="ref-8">8</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28743-fig-1.png"/>
</fig>
<p>Activity recognition in video sequences is a moving issue because of the comparability of visual substance, changes in the perspective for similar activities [<xref ref-type="bibr" rid="ref-9">9</xref>], camera movement with activity entertainer, posture and scale of an entertainer, and diverse enlightenment situations [<xref ref-type="bibr" rid="ref-10">10</xref>]. Human activities range from simple leg or arm movement to complex coordinated movement of consolidated legs, arms, and body. For example, kicking a football is a basic activity, whereas hopping for a top shoot is an aggregate movement of arms, legs, head, and entire body [<xref ref-type="bibr" rid="ref-11">11</xref>]. For many reasons, correctly recognition of human actions in video frames remains a difficult process, like having inter-class and intra-class variation, lightning, environmental and angle variation, etc. [<xref ref-type="bibr" rid="ref-12">12</xref>]. To deal with these issues handcrafted methods for feature extraction like histogram optical flow and histogram oriented gradient are used in previous research studies [<xref ref-type="bibr" rid="ref-13">13</xref>]. Because missing of a 3-dimensional (3D) structure in the video sequence, these methods are unable to recognize actions using 2D data [<xref ref-type="bibr" rid="ref-14">14</xref>].</p>
<p>Recently, the deep learning shows the much performance in the area of computer vision and machine learning for several applications such as medical [<xref ref-type="bibr" rid="ref-15">15</xref>], biometric [<xref ref-type="bibr" rid="ref-16">16</xref>], video surveillance, agriculture [<xref ref-type="bibr" rid="ref-17">17</xref>], and object classification [<xref ref-type="bibr" rid="ref-18">18</xref>]. From those, HAR is active research and many researchers improve the performance through deep learning techniques [<xref ref-type="bibr" rid="ref-19">19</xref>]. Convolutional neural network (CNN) is form of deep learning, being used to improve the rate of HAR [<xref ref-type="bibr" rid="ref-20">20</xref>]. Generally, HAR methods are based on two steps, i.e., features extraction and classification [<xref ref-type="bibr" rid="ref-21">21</xref>]. Different CNN pre-trained models like AlexNet [<xref ref-type="bibr" rid="ref-22">22</xref>], very very deep (VGG), and ResNet, and named a few others [<xref ref-type="bibr" rid="ref-23">23</xref>] are used with the transfer learning concept for HAR. These techniques give improved accuracy than traditional feature extraction techniques. But some time, due to complex nature of dataset, a single CNN model not performed well; therefore, information fusion of more than one model can be employed. The fusion process increases the computational time due to more number of predictors [<xref ref-type="bibr" rid="ref-24">24</xref>]. Hence, feature selection techniques are more suitable. The selection techniques minimize the volume of data to save the cost of modeling and, in some conditions, improve the functioning of the algorithm [<xref ref-type="bibr" rid="ref-9">9</xref>]. Finally, the final features are passed to the different classifiers for classification purposes. Different classifiers such as multiclass Support Vector Machine (M-SVM) [<xref ref-type="bibr" rid="ref-25">25</xref>], K-Nearest Neighbor (KNN), Linear Discriminant Analysis (LDA), Complex tree (CT), and Ada-boost are used for action classification [<xref ref-type="bibr" rid="ref-26">26</xref>]. Recently, the researchers introduced many techniques for human action recognition (HAR) [<xref ref-type="bibr" rid="ref-27">27</xref>]. Those techniques are based on a few well-known steps such as preprocessing of original video frames, region of interest detection (ROI) for more accurate features extraction, and finally recognition using machine learning classifiers [<xref ref-type="bibr" rid="ref-28">28</xref>].</p>
<p>However, they continue to face a slew of issues that degrade overall accuracy and lengthen computational time. The major challenges are as follows: (i) imbalanced datasets increase the prediction probability of the maximum class; (ii) feature extraction from the last layer does not correctly visualize the original human action due to a lack of the required number of features; and (iii) redundant and irrelevant features reduce system recognition accuracy. Furthermore, the presence of these features increases the computational time of the system. In this work, we proposed a new framework for HAR based on deep learning features fusion and improve whale optimization algorithm. Our major contributions are listed as follows:
<list list-type="bullet">
<list-item>
<p>Two pre-trained CNN models are fine-tuned, and new dense layers are added. The fine-tuned models were then trained on action datasets to extract features from a combination of layers (convolutional and fully connected) rather than a single target layer.</p></list-item>
<list-item>
<p>Using a modified correlation extended serial approach, the extracted features of both fine-tuned models are fused.</p></list-item>
<list-item>
<p>Based on the update criteria for the best feature selection, an improved whale optimization algorithm is introduced. Machine learning classifiers are used to classify the selected features.</p></list-item>
</list></p>
<p>The rest of the article is organized in the following order. Section 2 discussed the recent related work of HAR. Proposed HAR framework is discussed in Section 3. In this section the entire framework is described in the mathematical and visual manner. Results of the proposed HAR framework are presented in Section 4. The conclusion of this article is presented in Section 5.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Work</title>
<p>Many HAR techniques have been proposed in the literature based on deep learning and traditional features. Liu&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-29">29</xref>] introduced a two-stream deep neural network for HAR. This network recognizes unusual behavior of human in the video sequences. <bold>Volumetric motion history images</bold> (VMHI) and original frames are the two main parts of this model and tested on <bold>Royal Institute of Technology</bold> (KTH), Weizmann, and Ut-interaction datasets and showed improved accuracies. Chenarlogh&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-30">30</xref>] introduced three different CNN architectures to optimize the performance of HAR in limited data. Three architectures of this model include 1-stream, 2-stream, and 4-stream. They tested their architecture on the IXMAS dataset and attained average accuracy of 88.05% on 4-stream architecture. Sharif&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-31">31</xref>] suggested an approach to overcome the problem of the robust feature selection method. In HAR, extracting the prominent and salient features inside a video frame is a challenging job. The suggested method initially fuses three different feature categories and selects the most optimized features using strong correlation and Euclidean distance methods. Finally, classification is performed by a multi-class classifier. They used KTH, <bold>large human motion database</bold> (HMDB51), UCF YouTube, and Weizmann datasets and shows more than 94% classification accuracy. Jaouedi&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-32">32</xref>] introduced a hybrid deep network for HAR to overcome the problem of detecting a moving person from a scene and detecting human motion from a background. The suggested method is tested by using KTH, UCF101, and UCF sports datasets and attained an average accuracy of 96.3%. Sharif&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-33">33</xref>] suggested a novel HAR technique by using the combination of handcrafted and deep features. Initially, saliency-based method was employed for human silhouette extraction. Afterward, deep and handcrafted features are extracted and combined to make a final vector. The main purpose of features fusion is getting the maximum information of human actions for accurate classification. They tested their technique using UCF11 (YouTube), UT-interaction, IXMAS, Weizmann, and UCF sports datasets and attained better accuracy than SOTA.</p>
<p>Abdelbaky&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-34">34</xref>] presented an architecture PCANet TOP for feature extraction and action classification based on SVM classifier. They tested their method using UCF Sports, KTH, YouTube action, and Weizmann datasets and attained an accuracy of 92.67%. Afza&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-35">35</xref>] suggested a technique that fused traditional features and later selected the best of them for final classification. M-SVM classifier is used for action identification and achieved above 95% accuracy on four datasets- UCF YouTube, UCF Sports, Weizmann, and KTH. Abdelbaky&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-36">36</xref>] presented a simple Neural network based on (PCA) network to minimize the issues related to real-time recognition systems and 3-dimensional signals in a video frame. This scheme uses an unsupervised learning approach instead of supervised learning approach. Sahoo&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-37">37</xref>] suggested the HAR-Depth technique with shape learning and sequential learning streams combined with <bold>depth history image</bold> (DHI). The presented method is used to get maximum data from the action videos to overcome the error rate of correct recognition. Muhammad&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-38">38</xref>] suggested a <bold>Bi-Long shorter memory (BiLSTM)</bold> based HAR approach using Dilated Convolutional Neural Network. This approach gives better performance in video surveillance for security needs. The HAR sequential process was followed by the aforementioned techniques. They used CNN architectures to extract features but skipped the preprocessing and optimization steps. The difference between the above studies is the long computational time and redundant features that can be addressed by these two steps.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Proposed Methodology</title>
<p>The proposed HAR architecture is presenting in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>. The proposed framework includes several steps such as: (i) frames initial preprocessing (ii) fine-tuned two pre-trained deep learning models such as Inceptionv3 and Resnet101 and extract deep features (iii) fusion of deep learning features using modified correlation extended serial approach (iv) best features selection using improved whale optimization algorithm, and (v) classification using machine learning algorithms and compute results. The detail of each step is given in below subsections.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Proposed deep learning based framework for HAR</title>
</caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28743-fig-2.png"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Video Frames Preprocessing</title>
<p>Pre-processing is the most important steps in image processing, with applications in different fields like agriculture, medicine, and surveillance, to mention a few [<xref ref-type="bibr" rid="ref-39">39</xref>]. Pre-processing is critical in surveillance to deal with light changes, complicated backgrounds, noise reduction, and other issues. In this work, the pre-processing step is employed to convert the action video sequences into frames. Originally, the each extracted video frame having dimension <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mn>512</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>512</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mi>k</mml:mi></mml:math></inline-formula>, where <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula>. The converted frames are later resized into a size <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mn>256</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>256</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula>. These extracted video frames are converted into relevant classes and later utilized for the training of CNN models.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Convolutional Neural Network</title>
<p>CNN is a neural network with a convolution operation in at least one of its layers instead of matrix multiplications. CNN networks are now being used to improve the recognition rate of HAR. In a CNN, three basic layers are used: convolutional, pooling, and fully connected. In convolutional layer, different filters are applied to the image with different parameters for feature extraction. The basic parameters are size of kernel and the number of kernels. Mathematically, the convolution operation is formulated as follows:</p>
<p><disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mi>H</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">[</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>=</mml:mo><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mi>i</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mi>g</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>k</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">]</mml:mo></mml:math></disp-formula>where input image is denoted by <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mi>g</mml:mi></mml:math></inline-formula>, kernel by <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mi>i</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mi>a</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>b</mml:mi></mml:math></inline-formula> shows the row and column of the resultant matrix. The <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mo>&#x2217;</mml:mo></mml:math></inline-formula> represent the convolutional operator and <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mi>H</mml:mi></mml:math></inline-formula> represent the output of convolution operation. After each convolutional layer, a ReLu activation layer is added to remove the negative features and place with zero.</p>
<p><disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>L</mml:mi><mml:mi>u</mml:mi><mml:mo>=</mml:mo><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mi>x</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>H</mml:mi></mml:math></disp-formula></p>
<p>Pooling layer is used to decrease the size of the tensor to increase the calculation speed. In pooling layers, specific function is performed like max operation and average operation. Max pooling layer is used get a maximum value from each filter region and average pooling layer is utilized to get an AVG value in the each filter region. Another layer named fully connected layer, is employed to smooth the result before classification placed to output layer of a neural network. Mathematically, the FC layer is formulated as follows:</p>
<p><disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>H</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>b</mml:mi><mml:mo>]</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p><disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x2217;</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula></p>
<p><disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>where, <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is final FC layer, <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mi mathvariant="normal">&#x0394;</mml:mi></mml:math></inline-formula> represent activation function, and <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mi>i</mml:mi></mml:math></inline-formula> is layer number. After the FC layer, the Softmax classification layer is added for features classification.</p>
<p><disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mrow><mml:mi mathvariant="italic">S</mml:mi><mml:mi mathvariant="italic">o</mml:mi><mml:mi mathvariant="italic">f</mml:mi><mml:mi mathvariant="italic">t</mml:mi><mml:mi mathvariant="italic">m</mml:mi><mml:mi mathvariant="italic">a</mml:mi><mml:mi mathvariant="italic">x</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mi>F</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Deep Learning Features</title>
<p>In this work, two pre-trained CNN models namely Inceptionv3 [<xref ref-type="bibr" rid="ref-40">40</xref>] and Resnet101 [<xref ref-type="bibr" rid="ref-41">41</xref>] are utilized for features extraction. Inception V3 CNN consists of 01 input layer, 94 convolutional layers, 01 fully connected layer, and 04 MaxPooling layer. Total number of layers in this network are 315. This network accepts input image of size <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mn>229</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>299</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula>. ResNet101 CNN model comprises of 01 input layer, 105 convolutional layers, 01 fully connected layer, and 01 MaxPooling layer. The total numbers of layers in this network are 347. This network accepts the input image of size <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mn>224</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>224</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula>. Initially both models were trained on ImageNet dataset which have 1000 object classes. Therefore, we fine-tuned both models and removed the last layers (fully connected layer-Classification layer) and added new dense layers and trained on action datasets using transfer learning. We considered the 70% video frames for training and rest 30% for testing purposes. The transfer learning (TL) concepts opted for training of fine-tuned models. Since the pre-trained nets are trained on selective classes (i.e., ImageNet dataset) but in our case, the target task is action recognition. Therefore, we need to train the network on selected action dataset. In the case of InceptionV3 CNN model, the last three layers such as &#x2018;predictions&#x2019;, &#x2018;predictions_softmax&#x2019;, and &#x2018;ClassificationLayer_predictions&#x2019; are replaced with &#x2018;new_fc&#x2019;, &#x2018;predictions_softmax&#x2019;, and &#x2018;new_classoutput&#x2019; layers. In the case of Resnet101 the last three layers such as &#x2018;fc1000&#x2019;, &#x2018;prob&#x2019;, and &#x2018;ClassificationLayer_predictions&#x2019; are replaced with &#x2018;new_fc&#x2019;, &#x2018;prob&#x2019;, and &#x2018;new_classoutput&#x2019; layers. The hyper parameters are initialized such as mini batch size of 16, initial learning rate is 0.05, epochs 200, and dropout factor is 0.5. Then, the newly fine-tuned models are trained through TL. Visually, the process of TL is illustrated in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Process of TL for HAR</title>
</caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28743-fig-3.png"/>
</fig>
<p>Inception V3 Features: We use the avg_pool layer of fine-tuned Inception V3 CNN model and applied activation for features extraction. On this layer, a feature vector is obtained of dimension <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>2048</mml:mn></mml:math></inline-formula> and represented with <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>.</p>
<p>ResNet101 Features: We employed pool5 layer of fine-tuned ResNet101 model and applied activation function for feature extraction. On this layer, a feature vector of dimension <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>2048</mml:mn></mml:math></inline-formula> is obtained and represented with <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Deep Features Fusion</title>
<p>Features fusion is the process of combined multi-level information in one vector for better recognition accuracy. In this work, we fused two deep extracted feature vectors <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> using a new approach named modified correlation extended serial approach. Consider, <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, then the correlation is find out among <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mi>i</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:mi>j</mml:mi></mml:math></inline-formula> based on the following formula:</p>
<p><disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mover><mml:mi>M</mml:mi><mml:mo accent="false">&#x00AF;</mml:mo></mml:mover><mml:mo>)</mml:mo></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mover><mml:mi>N</mml:mi><mml:mo accent="false">&#x00AF;</mml:mo></mml:mover><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msqrt><mml:mo>&#x2211;</mml:mo><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mover><mml:mi>M</mml:mi><mml:mo accent="false">&#x00AF;</mml:mo></mml:mover><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x2211;</mml:mo><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mover><mml:mi>N</mml:mi><mml:mo accent="false">&#x00AF;</mml:mo></mml:mover><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:msqrt></mml:mfrac></mml:math></disp-formula></p>
<p>Based on this formula, the features that have positive correlation (&#x002B;1) are selected in a new vector denoted by <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and features that have correlation value 0 or &#x2212;1, are added in <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. After that, the mean value is computed of <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and compared each feature with that as follows:</p>
<p><disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mi>C</mml:mi><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover><mml:mi>V</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi>f</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2265;</mml:mo><mml:mi>&#x03BC;</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi mathvariant="italic">I</mml:mi><mml:mi mathvariant="italic">g</mml:mi><mml:mi mathvariant="italic">n</mml:mi><mml:mi mathvariant="italic">o</mml:mi><mml:mi mathvariant="italic">r</mml:mi><mml:mi mathvariant="italic">e</mml:mi></mml:mrow><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:mi mathvariant="italic">O</mml:mi><mml:mi mathvariant="italic">t</mml:mi><mml:mi mathvariant="italic">h</mml:mi><mml:mi mathvariant="italic">e</mml:mi><mml:mi mathvariant="italic">r</mml:mi><mml:mi mathvariant="italic">w</mml:mi><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">s</mml:mi><mml:mi mathvariant="italic">e</mml:mi></mml:mrow></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>The updated vector <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:msub><mml:mrow><mml:mover><mml:mi>V</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> are finally fused in one vector based on the following mathematical equation:</p>
<p><disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:msub><mml:mrow><mml:mtext>V</mml:mtext></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>k</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>k</mml:mtext></mml:mrow><mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mrow><mml:mtext>m</mml:mtext></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mrow><mml:mtext>n</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>V</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>k</mml:mtext></mml:mrow><mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mrow><mml:mtext>m</mml:mtext></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mrow><mml:mtext>n</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>The resultant feature vector obtained of dimension <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mi>N</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mrow><mml:mtext>V</mml:mtext></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, where the seize of <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:msub><mml:mrow><mml:mtext>V</mml:mtext></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is 2205 in this work that further optimized using improved whale optimization algorithm.</p>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Deep Features Optimization</title>
<p>For feature selection, we used an improved whale optimization algorithm (IWOA). The fused vector <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:msub><mml:mrow><mml:mtext>V</mml:mtext></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>k</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is given to IWOA as input and the algorithm returns the optimized feature vector as output. The working of optimization process is given below.</p>
<p><bold>Whale optimization algorithm</bold> (WOA) is a metaheuristic algorithm, presented by first Mirjalili in 2016 [<xref ref-type="bibr" rid="ref-42">42</xref>]. There are three basic steps performed in this algorithm namely encircling prey, spiral updating position, and random search for prey.</p>
<p>Encircling prey: The humpback whale will encircle the prey once the location of the prey has been established. The encircling prey mechanism of whale is formulated by <xref ref-type="disp-formula" rid="eqn-10">Eqs. (10)</xref> and <xref ref-type="disp-formula" rid="eqn-11">(11)</xref>.</p>
<p><disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mi>D</mml:mi><mml:mi>Y</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>Y</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p><disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:mi>Y</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>Y</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>B</mml:mi><mml:mi>E</mml:mi></mml:math></disp-formula>where <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>i</mml:mi></mml:math></inline-formula> is the current number of iterations; <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:mi>Y</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> denotes the best whale position vector by a long shot; <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:mi>Y</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> denotes the current whale position vector; <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mi>B</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:mi>D</mml:mi></mml:math></inline-formula> denote the vector coefficient and are calculated by <xref ref-type="disp-formula" rid="eqn-12">Eqs. (12)</xref> and <xref ref-type="disp-formula" rid="eqn-13">(13)</xref>.</p>
<p><disp-formula id="eqn-12"><label>(12)</label><mml:math id="mml-eqn-12" display="block"><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mi>b</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>b</mml:mi></mml:math></disp-formula></p>
<p><disp-formula id="eqn-13"><label>(13)</label><mml:math id="mml-eqn-13" display="block"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></disp-formula>where s<sub>1</sub> and s<sub>2</sub> denote the casual numbers <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:mrow><mml:mo>(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>; <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mi>b</mml:mi></mml:math></inline-formula> denotes a convergent factor and linearly decreased from 2 to 0; <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:mi>b</mml:mi></mml:math></inline-formula> is calculated by <xref ref-type="disp-formula" rid="eqn-14">Eq. (14)</xref>.</p>
<p><disp-formula id="eqn-14"><label>(14)</label><mml:math id="mml-eqn-14" display="block"><mml:mi>b</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mn>2</mml:mn><mml:mfrac><mml:mi>I</mml:mi><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mfrac></mml:math></disp-formula>where <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:mi>I</mml:mi></mml:math></inline-formula> denotes the current number of iterations and <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the maximum iterations.</p>
<p>Updating Spiral position: Because humpback whales swim in a circle toward their prey, therefore, the circular position updating is done through the following equation.</p>
<p><disp-formula id="eqn-15"><label>(15)</label><mml:math id="mml-eqn-15" display="block"><mml:mi>Y</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>Y</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msup><mml:mi>cos</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:mi>l</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>where <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mi>Y</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>Y</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:math></inline-formula> shows the distance between the prey and whale; <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:mi>b</mml:mi></mml:math></inline-formula> represents the constant and l stands for a unintended number from <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:mrow><mml:mo>(</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. Noticing that, while the whale go swimming in a curved toward its food, it also has to contract to envelop it. Therefore, the encircling prey method is selected by the probability <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and the circular model is chosen by <inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. <xref ref-type="disp-formula" rid="eqn-16">Eq. (16)</xref> illustrates the calculating procedure:</p>
<p><disp-formula id="eqn-16"><label>(16)</label><mml:math id="mml-eqn-16" display="block"><mml:mi>Y</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mi>Y</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>B</mml:mi><mml:mi>E</mml:mi></mml:mtd><mml:mtd><mml:mi>p</mml:mi><mml:mo>&#x003C;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>Y</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msup><mml:mi>cos</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:mi>l</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mi>p</mml:mi><mml:mo>&#x2265;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>It is set on the statistical method to attack prey and become close to prey in order to minimize the value of <inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:mi>b</mml:mi></mml:math></inline-formula>, so that <inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:msup><mml:mi>B</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mi>s</mml:mi></mml:math></inline-formula> range likewise fell with <inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:mi>b</mml:mi></mml:math></inline-formula> in the iteration progression. When the value of <inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:mi>b</mml:mi></mml:math></inline-formula> falls from 2 to 0, <inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:mi>B</mml:mi></mml:math></inline-formula> is said to be within a random value [&#x2212;b, b]. Furthermore, when the value of <inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:mi>A</mml:mi></mml:math></inline-formula> is [&#x2212;1, 1], the whale&#x0027;s next place could be right now or anywhere else between its prey. The whale attacks its victim when <inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:mi>B</mml:mi></mml:math></inline-formula> is smaller than 1. While swimming along with the spiral pattern, the humpback whale surrounds its prey. To replicate the whale&#x2019;s hunting behavior, the probability of the encircling prey mechanism and curl position revise is set to 0.3.</p>
<p>Arbitrary search for prey: When a whale goes on randomly searching for prey, it must vary its position by going on a random search. The positions are computed as follows:</p>
<p><disp-formula id="eqn-17"><label>(17)</label><mml:math id="mml-eqn-17" display="block"><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mi>D</mml:mi><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>Y</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p><disp-formula id="eqn-18"><label>(18)</label><mml:math id="mml-eqn-18" display="block"><mml:mi>Y</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mi>B</mml:mi><mml:mi>E</mml:mi></mml:math></disp-formula>where <inline-formula id="ieqn-54"><mml:math id="mml-ieqn-54"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the casually choosing the whale&#x2019;s position vector. When <inline-formula id="ieqn-55"><mml:math id="mml-ieqn-55"><mml:mi>B</mml:mi><mml:mo>&#x2265;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>, a seeking agent will refresh the positions of all other whales to the searching whale, forcing it to flee the target in order to locate better feed. In this approach, the exploration ability of the algorithm may be improved, allowing WOA to be searched from all angles.</p>
<p>In this work, we update two primary parameters <inline-formula id="ieqn-56"><mml:math id="mml-ieqn-56"><mml:mi>B</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-57"><mml:math id="mml-ieqn-57"><mml:mi>D</mml:mi></mml:math></inline-formula>. This algorithm can balance exploitation and exploration because to its <inline-formula id="ieqn-58"><mml:math id="mml-ieqn-58"><mml:mi>B</mml:mi></mml:math></inline-formula> set. As a result, the likelihood of a locally optimal increases. The <inline-formula id="ieqn-59"><mml:math id="mml-ieqn-59"><mml:mi>B</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-60"><mml:math id="mml-ieqn-60"><mml:mi>D</mml:mi></mml:math></inline-formula> parameters in WOA are set to 0.4 and 0.5, respectively, which is clearly unnecessary. Meanwhile, WOA&#x0027;s capacity to search for all-around optimization has to be improved. The revised definition is defined as follows:</p>
<p><disp-formula id="eqn-19"><label>(19)</label><mml:math id="mml-eqn-19" display="block"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2217;</mml:mo><mml:mi>b</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi>i</mml:mi><mml:mo>&#x003E;</mml:mo><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math></disp-formula></p>
<p><disp-formula id="eqn-20"><label>(20)</label><mml:math id="mml-eqn-20" display="block"><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mi></mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula></p>
<p><disp-formula id="eqn-21"><label>(21)</label><mml:math id="mml-eqn-21" display="block"><mml:mi>B</mml:mi><mml:mo>=</mml:mo><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mn>30</mml:mn><mml:mi>x</mml:mi><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mi>i</mml:mi><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msup><mml:mo>]</mml:mo></mml:mrow></mml:math></disp-formula>where <inline-formula id="ieqn-61"><mml:math id="mml-ieqn-61"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is the initial probability of adaptive search enclosing mechanism; <inline-formula id="ieqn-62"><mml:math id="mml-ieqn-62"><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula id="ieqn-63"><mml:math id="mml-ieqn-63"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> are the probability of encircling prey mechanism of <inline-formula id="ieqn-64"><mml:math id="mml-ieqn-64"><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-65"><mml:math id="mml-ieqn-65"><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:math></inline-formula> generation; <inline-formula id="ieqn-66"><mml:math id="mml-ieqn-66"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> shows the probability of minimum enveloping; <inline-formula id="ieqn-67"><mml:math id="mml-ieqn-67"><mml:msubsup><mml:mi>P</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> denotes the probability of updated helix position of ith generation; <inline-formula id="ieqn-68"><mml:math id="mml-ieqn-68"><mml:mi>i</mml:mi></mml:math></inline-formula> shows the iterations and <inline-formula id="ieqn-69"><mml:math id="mml-ieqn-69"><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the maximum iteration and <inline-formula id="ieqn-70"><mml:math id="mml-ieqn-70"><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:math></inline-formula>.</p>
<p>The jumping behavior is also changed when the whale attempts to divide the region the value of local optimal can drop into minimum value by randomly updating the whale&#x2019;s location. The jumping behavior is defined as follows:</p>
<p><disp-formula id="eqn-22"><label>(22)</label><mml:math id="mml-eqn-22" display="block"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mn>2</mml:mn><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>d</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:math></disp-formula>where <inline-formula id="ieqn-71"><mml:math id="mml-ieqn-71"><mml:mi>g</mml:mi></mml:math></inline-formula> denotes jumping coefficient and <inline-formula id="ieqn-72"><mml:math id="mml-ieqn-72"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes all whales. The neural network is employed as a fitness function and the fitness is calculated based on the mean square error rate (MSER). The best selected features are passed to several machine learning classifiers for the final action recognition.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experimental Results and Discussion</title>
<p>This section comprises a full discussion of the results and analyses. The proposed framework is tested on four different datasets namely, (i) Ut-interaction, (ii) UCF Sports, (iii) Hollywood, and (iv) IXMAS. Several hyperparameters are employed for the training of pre-trained models such as learning rate is 0.05, mini batch size is 16, epochs are 200, and optimizer is stochastic gradient descent. The 10 fold cross-validation is opted on all 4 datasets, where the training and testing ratio was 50:50. Eight different classifiers, including Fine <bold>K-nearest neighbour</bold> (KNN), Ensemble Subspace KNN, Cubic SVM, Weighted KNN, Linear SVM, Cosine KNN, Quadratic SVM, Medium KNN, and Ensemble Bagged Trees are utilized for the classification results. The proposed framework is implemented in MATLAB 2020a, using personal computer having specification, Core i7 with 16 GB of DDR4 RAM and 16GB graphics card.</p>
<sec id="s4_1">
<label>4.1</label>
<title>Numerical Results</title>
<p>The proposed framework results are presenting here in the form of tabular and confusion matrixes. The results are presented here for each dataset separately.</p>
<p>UT-Interaction Dataset Results: The results of UT-Interaction dataset are presented in <xref ref-type="table" rid="table-1">Tab. 1</xref>. The <bold>Ensemble Subspace KNN</bold> (ESKNN) classifier attained the highest accuracy of 100% and other parameters like precision, recall, and F1 score values are 1.0, 1.0, and 1.0, respectively. The rest of the classifiers also achieved better results of &#x003E;99%. <xref ref-type="fig" rid="fig-4">Fig. 4</xref> illustrated the confusion matrix of ESKNN classifier. Through this figure, the computed performance measures can be verified. The computational time of each classifier is also noted and the minimum testing time is 58.508 (s) of Ensemble Baggage Tree.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Classification results on UT-Interaction dataset using proposed framework</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th rowspan="2">Classifiers</th>
<th align="center" colspan="5">Parameters</th>
</tr>
<tr>
<th>Recall</th>
<th>Precision</th>
<th>F1 Score</th>
<th>Accuracy (%)</th>
<th>Time (s)</th>
</tr>
</thead>
<tbody>
<tr>
<td><bold>ES KNN</bold></td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td><bold>100</bold></td>
<td>171</td>
</tr>
<tr>
<td>Fine KNN</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>99.9</td>
<td>121.9</td>
</tr>
<tr>
<td>Cubic SVM</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>99.9</td>
<td>100.08</td>
</tr>
<tr>
<td>Weighted KNN</td>
<td>0.9983</td>
<td>0.9983</td>
<td>0.9966</td>
<td>99.7</td>
<td>104.95</td>
</tr>
<tr>
<td>Cosine KNN</td>
<td>0.9937</td>
<td>0.99</td>
<td>0.9937</td>
<td>99.4</td>
<td>101.02</td>
</tr>
<tr>
<td>Quadratic SVM</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>99.9</td>
<td>92.417</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>0.9983</td>
<td>0.9983</td>
<td>0.9966</td>
<td>99.7</td>
<td>80.563</td>
</tr>
<tr>
<td>Medium KNN</td>
<td>0.9916</td>
<td>0.9961</td>
<td>0.9916</td>
<td>99.2</td>
<td>124.71</td>
</tr>
<tr>
<td>EBT</td>
<td>0.9816</td>
<td>0.98</td>
<td>0.9783</td>
<td>98.3</td>
<td><bold>58.508</bold></td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Subspace KNN classifier&#x2019;s confusion matrix on Ut-Interaction dataset using optimal features fusion</title>
</caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28743-fig-4.png"/>
</fig>
<p>UCF Sports Dataset Results: <xref ref-type="table" rid="table-2">Tab. 2</xref> presents the recognition results of UCF Sports dataset using proposed framework. In this table, Quadratic SVM classifier attained the highest accuracy of 100% and other parameters like precision, recall, and F1 score values are 1.0, 1.0, and 1.0, respectively. These values can be further verified through a confusion matrix given in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>. The other classifiers also give the better accuracy using proposed framework on selected dataset. The computational time is also noted for each classifier and minimum noted time is 107.02 (s) of Ensemble Baggage Tree (EBT). Similarly, the hollywood dataset results are presented in <xref ref-type="table" rid="table-3">Tab. 3</xref> and confusion matrix illustrated in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Classification results on UCF Sports dataset using proposed framework</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th rowspan="2">Classifiers</th>
<th align="center" colspan="5">Parameters</th>
</tr>
<tr>
<th>Recall</th>
<th>Precision</th>
<th>F1 Score</th>
<th>Accuracy (%)</th>
<th>Time (s)</th>
</tr>
</thead>
<tbody>
<tr>
<td>ES KNN</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>100</td>
<td>241.5</td>
</tr>
<tr>
<td>Fine KNN</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>100</td>
<td>178.21</td>
</tr>
<tr>
<td>Cubic SVM</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>100</td>
<td>178.66</td>
</tr>
<tr>
<td>Weighted KNN</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>99.9</td>
<td>168.68</td>
</tr>
<tr>
<td>Cosine KNN</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>99.9</td>
<td>161.11</td>
</tr>
<tr>
<td><bold>Quadratic SVM</bold></td>
<td><bold>1.0</bold></td>
<td><bold>1.0</bold></td>
<td><bold>1.0</bold></td>
<td><bold>100</bold></td>
<td>174.86</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>99.9</td>
<td>144.5</td>
</tr>
<tr>
<td>Medium KNN</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>99.9</td>
<td>171.76</td>
</tr>
<tr>
<td>EBT</td>
<td>0.9937</td>
<td>0.99</td>
<td>0.9925</td>
<td>99.3</td>
<td><bold>107.02</bold></td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Quadratic SVM classifier&#x2019;s confusion matrix on UCF Sports dataset using optimal features fusion</title>
</caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28743-fig-5.png"/>
</fig><table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Classification results on Hollywood dataset using proposed framework</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th rowspan="2">Classifier</th>
<th align="center" colspan="5">Parameters</th>
</tr>
<tr>
<th>Recall</th>
<th>Precision</th>
<th>F1 Score</th>
<th>Accuracy (%)</th>
<th>Time (s)</th>
</tr>
</thead>
<tbody>
<tr>
<td>Ensemble<break/>Subspace KNN</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td><bold>99.9</bold></td>
<td>139.4</td>
</tr>
<tr>
<td>Fine KNN</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>99.9</td>
<td>211.5</td>
</tr>
<tr>
<td>Cubic SVM</td>
<td>1.0</td>
<td>1.0</td>
<td>1.0</td>
<td>99.8</td>
<td>150.6</td>
</tr>
<tr>
<td>Weighted KNN</td>
<td>0.9983</td>
<td>0.9983</td>
<td>0.9966</td>
<td>99.7</td>
<td>210.5</td>
</tr>
<tr>
<td>Cosine KNN</td>
<td>0.9966</td>
<td>0.9958</td>
<td>0.9966</td>
<td>99.5</td>
<td>194.4</td>
</tr>
<tr>
<td>Quadratic SVM</td>
<td>0.9983</td>
<td>0.9983</td>
<td>0.9966</td>
<td>99.7</td>
<td>147.8</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>0.9937</td>
<td>0.99</td>
<td>0.9925</td>
<td>99.3</td>
<td>278.3</td>
</tr>
<tr>
<td>Medium KNN</td>
<td>0.9937</td>
<td>0.99</td>
<td>0.9937</td>
<td>99.4</td>
<td>216.2</td>
</tr>
<tr>
<td>Ensemble Bagged Trees</td>
<td>0.9887</td>
<td>0.9812</td>
<td>0.9837</td>
<td>98.7</td>
<td><bold>120.8</bold></td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Fine KNN classifier&#x2019;s confusion matrix on Hollywood dataset using optimal features fusion</title>
</caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28743-fig-6.png"/>
</fig>
<p>IXMAS Dataset Results: The results of IXMAS action dataset are given in <xref ref-type="table" rid="table-4">Tab. 4</xref> using proposed framework. This table shows the best accuracy is achieved by ESKNN classifier of 99.1% and other measures like precision, recall, and F1 score are 0.9916, 0.9916, and 0.99, respectively. These values can be further verified through a confusion matrix, illustrated in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>. Foe each classifier listed in this table, the computational time is also computed. The minimum noted time is 178.1 (s) for LSVM, whereas the EBT executed in 202.87 (s). Overall, the ESKNN classifier performs better than the rest on the classifier based on time and accuracy.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Classification results on IXMAS dataset using proposed framework</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th rowspan="2">Classifier</th>
<th align="center" colspan="5">Parameters</th>
</tr>
<tr>
<th>Recall</th>
<th>Precision</th>
<th>F1 Score</th>
<th>Accuracy (%)</th>
<th>Time (s)</th>
</tr>
</thead>
<tbody>
<tr>
<td>Ensemble<break/>Subspace KNN</td>
<td>0.9916</td>
<td>0.9916</td>
<td>0.99</td>
<td><bold>99.1</bold></td>
<td>492.8</td>
</tr>
<tr>
<td>Fine KNN</td>
<td>0.9909</td>
<td>0.9909</td>
<td>0.99</td>
<td>99.0</td>
<td>291.9</td>
</tr>
<tr>
<td>Cubic SVM</td>
<td>0.9866</td>
<td>0.985</td>
<td>0.985</td>
<td>98.7</td>
<td>339.8</td>
</tr>
<tr>
<td>Weighted KNN</td>
<td>0.9716</td>
<td>0.97</td>
<td>0.9716</td>
<td>97.5</td>
<td>295.5</td>
</tr>
<tr>
<td>Cosine KNN</td>
<td>0.9541</td>
<td>0.955</td>
<td>0.9558</td>
<td>95.8</td>
<td>242.2</td>
</tr>
<tr>
<td>Quadratic SVM</td>
<td>0.9825</td>
<td>0.9808</td>
<td>0.98</td>
<td>98.2</td>
<td>255.7</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>0.9683</td>
<td>0.9633</td>
<td>0.965</td>
<td>96.8</td>
<td><bold>178.1</bold></td>
</tr>
<tr>
<td>Medium KNN</td>
<td>0.9612</td>
<td>0.966</td>
<td>0.9652</td>
<td>95.9</td>
<td>293.6</td>
</tr>
<tr>
<td>Ensemble Bagged Trees</td>
<td>0.9525</td>
<td>0.9533</td>
<td>0.9541</td>
<td>95.0</td>
<td>202.87</td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>KNN classifier&#x2019;s confusion matrix on IXMAS dataset using optimal features fusion</title>
</caption>
<graphic mimetype="image" mime-subtype="png" xlink:href="CMC_28743-fig-7.png"/>
</fig>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Discussion</title>
<p>A detailed discussion has been conducted in this section for the proposed framework based on the accuracy achieved by the selected datasets. The proposed framework is illustrated in <xref ref-type="fig" rid="fig-2">Fig. 2</xref> which consists of series of steps. The entire proposed framework results are given in <xref ref-type="table" rid="table-1">Tabs. 1</xref>&#x2013;<xref ref-type="table" rid="table-4">4</xref> and confusion matrixes in <xref ref-type="fig" rid="fig-4">Figs. 4</xref>&#x2013;<xref ref-type="fig" rid="fig-7">7</xref>. Based on the tables and confusion matrixes, it is noted the proposed framework achieved maximum accuracy on selected datasets. However, it is essential to analyse the performance of middle steps such as original deep features extraction and best selected features for each CNN model.</p>

<p>The main purpose of employing classification results of middle steps is to analyse the importance of optimization algorithm. Another purpose of this analysis is to check the following question: if optimization algorithm is employed separately on deep extracted features then what will be the accuracy?</p>
<p><xref ref-type="table" rid="table-5">Tabs. 5</xref>&#x2013;<xref ref-type="table" rid="table-8">8</xref> presents the accuracy of middle steps on selected action dataset. In these tables, it is noted that the accuracy is initially computed by using fine-tuned ResNet101 and Inception V3 models features. After that, the optimization algorithm is employed on original deep extracted features of ResNet101 (Best ResNet101) and Inception V3 (Best Inception V3). The proposed framework results are given in the last column for the sake of comparison. Based on accuracy values, given in these tables, it is noted that the optimization process improves the recognition accuracy but one the other end, proposed framework gives the better results.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Comparison of overall proposed framework accuracy with middle steps on UT-Interaction dataset</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Classifiers</th>
<th>ResNet101</th>
<th>Inception V3</th>
<th>Best ResNet101</th>
<th>Best Inception V3</th>
<th>Proposed</th>
</tr>
</thead>
<tbody>
<tr>
<td>Ensemble Subspace KNN</td>
<td>95.2</td>
<td>96.5</td>
<td>98.8</td>
<td>96.6</td>
<td><bold>100</bold></td>
</tr>
<tr>
<td><bold>Fine KNN</bold></td>
<td>96.1</td>
<td>95.6</td>
<td>98.9</td>
<td>97.6</td>
<td>99.9</td>
</tr>
<tr>
<td>Cubic SVM</td>
<td>96.5</td>
<td>95.5</td>
<td>97.7</td>
<td>96.6</td>
<td>99.9</td>
</tr>
<tr>
<td>Weighted KNN</td>
<td>94.8</td>
<td>96.9</td>
<td>97.2</td>
<td>97.9</td>
<td>99.7</td>
</tr>
<tr>
<td>Cosine KNN</td>
<td>96</td>
<td>97.2</td>
<td>96.7</td>
<td>98.2</td>
<td>99.4</td>
</tr>
<tr>
<td>Quadratic SVM</td>
<td>97</td>
<td>96.2</td>
<td>98.6</td>
<td>98.5</td>
<td>99.9</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>96.9</td>
<td>95.5</td>
<td>97.4</td>
<td>98.2</td>
<td>99.7</td>
</tr>
<tr>
<td>Medium KNN</td>
<td>97</td>
<td>94.1</td>
<td>96.5</td>
<td>98.1</td>
<td>99.2</td>
</tr>
<tr>
<td>Ensemble Bagged Trees</td>
<td>95.7</td>
<td>95.6</td>
<td>97.1</td>
<td>97.9</td>
<td>98.3</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>Comparison of overall proposed framework accuracy with middle steps on UCF Sports dataset</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Classifiers</th>
<th>ResNet101</th>
<th>Inception V3</th>
<th>Best ResNet101</th>
<th>Best Inception V3</th>
<th>Proposed</th>
</tr>
</thead>
<tbody>
<tr>
<td>Ensemble Subspace KNN</td>
<td>94.6</td>
<td>94.9</td>
<td>95.8</td>
<td>95.5</td>
<td>100</td>
</tr>
<tr>
<td>Fine KNN</td>
<td>95.8</td>
<td>93.9</td>
<td>96.7</td>
<td>95.3</td>
<td>100</td>
</tr>
<tr>
<td>Cubic SVM</td>
<td>93.7</td>
<td>94.9</td>
<td>94.1</td>
<td>95.6</td>
<td>100</td>
</tr>
<tr>
<td>Weighted KNN</td>
<td>93.6</td>
<td>92.8</td>
<td>95.5</td>
<td>95.0</td>
<td>99.9</td>
</tr>
<tr>
<td>Cosine KNN</td>
<td>94.5</td>
<td>95.8</td>
<td>95.3</td>
<td>96.4</td>
<td>99.9</td>
</tr>
<tr>
<td>Quadratic SVM</td>
<td>94.8</td>
<td>94.9</td>
<td>95.9</td>
<td>96.0</td>
<td>100</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>94.1</td>
<td>95.8</td>
<td>96.4</td>
<td>96.8</td>
<td>99.9</td>
</tr>
<tr>
<td>Medium KNN</td>
<td>93.5</td>
<td>94.1</td>
<td>97.8</td>
<td>95.7</td>
<td>99.9</td>
</tr>
<tr>
<td>Ensemble Bagged Trees</td>
<td>94.2</td>
<td>93.6</td>
<td>98.1</td>
<td>94.8</td>
<td>99.3</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-7">
<label>Table 7</label>
<caption>
<title>Comparison of overall proposed framework accuracy with middle steps on Hollywood dataset</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Classifiers</th>
<th>ResNet101</th>
<th>Inception V3</th>
<th>Best ResNet101</th>
<th>Best Inception V3</th>
<th>Proposed</th>
</tr>
</thead>
<tbody>
<tr>
<td>Ensemble Subspace KNN</td>
<td>95.7</td>
<td>95.8</td>
<td>96.8</td>
<td>97.3</td>
<td>99.9</td>
</tr>
<tr>
<td>Fine KNN</td>
<td>95.9</td>
<td>97.1</td>
<td>97.9</td>
<td>98.0</td>
<td>99.9</td>
</tr>
<tr>
<td>Cubic SVM</td>
<td>95.8</td>
<td>96.5</td>
<td>96.5</td>
<td>97.4</td>
<td>99.8</td>
</tr>
<tr>
<td>Weighted KNN</td>
<td>95.5</td>
<td>94.4</td>
<td>97.1</td>
<td>96.3</td>
<td>99.7</td>
</tr>
<tr>
<td>Cosine KNN</td>
<td>96.1</td>
<td>96.7</td>
<td>96.5</td>
<td>97.7</td>
<td>99.5</td>
</tr>
<tr>
<td>Quadratic SVM</td>
<td>96.7</td>
<td>95.9</td>
<td>97.1</td>
<td>96.9</td>
<td>99.7</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>95.3</td>
<td>96.2</td>
<td>95.4</td>
<td>97.5</td>
<td>99.3</td>
</tr>
<tr>
<td>Medium KNN</td>
<td>95.0</td>
<td>95.6</td>
<td>97.8</td>
<td>98.4</td>
<td>99.4</td>
</tr>
<tr>
<td>Ensemble Bagged Trees</td>
<td>93.8</td>
<td>94.2</td>
<td>97.3</td>
<td>96.0</td>
<td>98.7</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-8">
<label>Table 8</label>
<caption>
<title>Comparison of overall proposed framework accuracy with middle steps on IXMAS dataset</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Classifiers</th>
<th>ResNet101</th>
<th>Inception V3</th>
<th>Best ResNet101</th>
<th>Best Inception V3</th>
<th>Proposed</th>
</tr>
</thead>
<tbody>
<tr>
<td>ES KNN</td>
<td>94.2</td>
<td>94.9</td>
<td>95.7</td>
<td>96.7</td>
<td>99.1</td>
</tr>
<tr>
<td>Fine KNN</td>
<td>93.3</td>
<td>93.4</td>
<td>94.8</td>
<td>95.4</td>
<td>99.0</td>
</tr>
<tr>
<td>Cubic SVM</td>
<td>92.5</td>
<td>93.4</td>
<td>93.6</td>
<td>95.4</td>
<td>98.7</td>
</tr>
<tr>
<td>Weighted KNN</td>
<td>93.5</td>
<td>94.7</td>
<td>94.6</td>
<td>95.6</td>
<td>97.5</td>
</tr>
<tr>
<td>Cosine KNN</td>
<td>92.9</td>
<td>93.6</td>
<td>93.5</td>
<td>94.9</td>
<td>95.8</td>
</tr>
<tr>
<td>Quadratic SVM</td>
<td>91.1</td>
<td>92.5</td>
<td>92.2</td>
<td>96.2</td>
<td>98.2</td>
</tr>
<tr>
<td>Linear SVM</td>
<td>92.6</td>
<td>91.3</td>
<td>93.7</td>
<td>93.7</td>
<td>96.8</td>
</tr>
<tr>
<td>Medium KNN</td>
<td>92.8</td>
<td>93.1</td>
<td>93.9</td>
<td>95.4</td>
<td>95.9</td>
</tr>
<tr>
<td>EBT</td>
<td>91.5</td>
<td>94.4</td>
<td>92.1</td>
<td>95.8</td>
<td>95.0</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>At the end, a comparison of proposed framework accuracy is conducted with state of the art (SOTA) techniques using different selected datasets, as given in <xref ref-type="table" rid="table-9">Tab. 9</xref>. In this table, it is noted that authors of [<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>,<xref ref-type="bibr" rid="ref-43">43</xref>] used UCF sports dataset and achieved accuracies of 99.3%, 96.8%, and 92.67%, respectively. The proposed method attained 100% on UCF Sports dataset with minimum execution time. Similarly, authors used UT-Interaction dataset and achieved accuracies of 96.7%, 96.4%, and 99%. The proposed method achieved an accuracy of 100%. For IXMAS and Hollywood dataset, the proposed framework achieved an accuracy of 99.1% and 99.9% which is improved than the recent methods. Overall, values given in this table, it is clear that the proposed framework of HAR achieved improved accuracy than SOTA techniques.</p>
<table-wrap id="table-9">
<label>Table 9</label>
<caption>
<title>Comparison of proposed framework with SOTA techniques</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Reference</th>
<th>Year</th>
<th>Dataset</th>
<th>Accuracy (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td>[<xref ref-type="bibr" rid="ref-43">43</xref>]</td>
<td>2021</td>
<td>UCF Sports</td>
<td>96.8</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-35">35</xref>]</td>
<td>2021</td>
<td>UCF Sports</td>
<td>99.3</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-34">34</xref>]</td>
<td>2021</td>
<td>UCF Sports</td>
<td>92.67</td>
</tr>
<tr>
<td><bold>Proposed</bold></td>
<td><bold>-</bold></td>
<td><bold>UCF Sports</bold></td>
<td><bold>100</bold></td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-44">44</xref>]</td>
<td>2021</td>
<td>UT-Interaction</td>
<td>96.7</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-45">45</xref>]</td>
<td>2021</td>
<td>UT-Interaction</td>
<td>96.4</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-29">29</xref>]</td>
<td>2021</td>
<td>UT-Interaction</td>
<td>99</td>
</tr>
<tr>
<td><bold>Proposed</bold></td>
<td><bold>-</bold></td>
<td><bold>UT-Interaction</bold></td>
<td><bold>100</bold></td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-21">21</xref>]</td>
<td>2020</td>
<td>IXMAS</td>
<td>95.2</td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-30">30</xref>]</td>
<td>2019</td>
<td>IXMAS</td>
<td>88.05</td>
</tr>
<tr>
<td><bold>Proposed</bold></td>
<td><bold>-</bold></td>
<td><bold>IXMAS</bold></td>
<td><bold>99.1</bold></td>
</tr>
<tr>
<td>[<xref ref-type="bibr" rid="ref-6">6</xref>]</td>
<td>2021</td>
<td>Hollywood</td>
<td>99.2</td>
</tr>
<tr>
<td><bold>Proposed</bold></td>
<td><bold>-</bold></td>
<td><bold>Hollywood</bold></td>
<td><bold>99.9</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusion</title>
<p>Human action recognition (HAR) is rapidly gaining popularity in the field of pattern recognition and machine learning based on its important application-video surveillance. In this article, a new framework is proposed for HAR based deep learning and improved WOA. The experimental process is conducted on four publicly accessible datasets such as Ut-Interaction, Hollywood, IXMAS, and UCF Sports and attained an accuracy of 100%, 99.9%, 99.1%, and 100%, respectively. Comparison with SOTA techniques, it is observed that the proposed framework recognition accuracy is improved than the recent techniques. From the results, we conclude that the fusion based framework give the better accuracy than recognition performance on individual deep learning features and optimization algorithm. The optimization algorithm reduces the execution time during the testing process. The improved optimization algorithm reduced computational time without reducing classification accuracy, which is the work&#x0027;s strength. In the future, large datasets such as UCF101, Muhavi, and HMDB51 will be used for evaluation. Furthermore, for HAR, a single stream CNN framework will be considered.</p>
</sec>
</body>
<back>
<fn-group>
<fn fn-type="other"><p><bold>Funding Statement:</bold> The authors would like to thank research network for collaboration. This research work is supported in part by Chiang Mai University and HITEC University.</p>
</fn>
<fn fn-type="conflict"><p><bold>Conflicts of Interest:</bold> The authors declare that they have no conflicts of interest to report regarding the present study.</p>
</fn>
</fn-group>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Sharif</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Akram</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Raza</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Saba</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Rehman</surname></string-name></person-group>, &#x201C;<article-title>Hand-crafted and deep convolutional neural network features fusion and selection strategy: An application to intelligent human action recognition</article-title>,&#x201D; <source>Applied Soft Computing</source>, vol. <volume>87</volume>, no. <issue>2</issue>, pp. <fpage>105986</fpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y. D.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Allison</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Kadry</surname></string-name>, <string-name><given-names>S. H.</given-names> <surname>Wang</surname></string-name> and <string-name><given-names>T.</given-names> <surname>Saba</surname></string-name></person-group>, &#x201C;<article-title>A fused heterogeneous deep neural network and robust feature selection framework for human actions recognition</article-title>,&#x201D; <source>Arabian Journal for Science and Engineering</source>, vol. <volume>11</volume>, no. <issue>2</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>16</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>P.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Lan</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Xing</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Zeng</surname></string-name> and <string-name><given-names>N.</given-names> <surname>Zheng</surname></string-name></person-group>, &#x201C;<article-title>View adaptive neural networks for high performance skeleton-based human action recognition</article-title>,&#x201D; <source>IEEE Transactions on Pattern Analysis and Machine Intelligence</source>, vol. <volume>41</volume>, no. <issue>4</issue>, pp. <fpage>1963</fpage>&#x2013;<lpage>1978</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Zhao</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Xu</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Su</surname></string-name> and <string-name><given-names>Q.</given-names> <surname>Ji</surname></string-name></person-group>, &#x201C;<article-title>Bayesian hierarchical dynamic model for human action recognition</article-title>,&#x201D; in <conf-name>Proc. of the IEEE/CVF Conf. on Computer Vision and Pattern Recognition</conf-name>, <publisher-loc>NY, USA</publisher-loc>, pp. <fpage>7733</fpage>&#x2013;<lpage>7742</lpage>, <year>2019</year>. </mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Kamel</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Sheng</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Yang</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Li</surname></string-name> and <string-name><given-names>D. D.</given-names> <surname>Feng</surname></string-name></person-group>, &#x201C;<article-title>Deep convolutional neural networks for human action recognition using depth maps and postures</article-title>,&#x201D; <source>IEEE Transactions on Systems, Man, and Cybernetics: Systems</source>, vol. <volume>49</volume>, no. <issue>6</issue>, pp. <fpage>1806</fpage>&#x2013;<lpage>1819</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Khan</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Alhaisoni</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Tariq</surname></string-name>, <string-name><given-names>H. S.</given-names> <surname>Yong</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Armghan</surname></string-name></person-group>, &#x201C;<article-title>Human action recognition: A paradigm of best deep learning features selection and serial based extended fusion</article-title>,&#x201D; <source>Sensors</source>, vol. <volume>21</volume>, no. <issue>11</issue>, pp. <fpage>7941</fpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y. D.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>S. A.</given-names> <surname>Khan</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Attique</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Rehman</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Seo</surname></string-name></person-group>, &#x201C;<article-title>A resource conscious human action recognition framework using 26-layered deep convolutional neural network</article-title>,&#x201D; <source>Multimedia Tools and Applications</source>, vol. <volume>80</volume>, no. <issue>4</issue>, pp. <fpage>35827</fpage>&#x2013;<lpage>35849</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Slimani</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Benezeth</surname></string-name> and <string-name><given-names>F.</given-names> <surname>Souami</surname></string-name></person-group>, &#x201C;<article-title>Learning bag of spatio-temporal features for human interaction recognition</article-title>,&#x201D; in <conf-name>Twelfth Int. Conf. on Machine Vision</conf-name>, <publisher-loc>New Delhi, India</publisher-loc>, pp. <fpage>1143302</fpage>, <year>2020</year>. </mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Sharif</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Zahid</surname></string-name>, <string-name><given-names>J. H.</given-names> <surname>Shah</surname></string-name> and <string-name><given-names>T.</given-names> <surname>Akram</surname></string-name></person-group>, &#x201C;<article-title>Human action recognition: A framework of statistical weighted segmentation and rank correlation-based selection</article-title>,&#x201D; <source>Pattern Analysis and Applications</source>, vol. <volume>23</volume>, no. <issue>8</issue>, pp. <fpage>281</fpage>&#x2013;<lpage>294</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Ahmed</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Ramzan</surname></string-name>, <string-name><given-names>H. U.</given-names> <surname>Khan</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Iqbal</surname></string-name> and <string-name><given-names>J. I.</given-names> <surname>Choi</surname></string-name></person-group>, &#x201C;<article-title>Real-time violent action recognition using key frames extraction and deep learning</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>69</volume>, no. <issue>2</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>15</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>I. M.</given-names> <surname>Nasir</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Raza</surname></string-name>, <string-name><given-names>J. H.</given-names> <surname>Shah</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Rehman</surname></string-name></person-group>, &#x201C;<article-title>Human action recognition using machine learning in uncontrolled environment</article-title>,&#x201D; in <conf-name>2021 1st Int. Conf. on Artificial Intelligence and Data Analytics</conf-name>, <publisher-loc>Riydah, Saudi Arabia</publisher-loc>, pp. <fpage>182</fpage>&#x2013;<lpage>187</lpage>, <year>2021</year>. </mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Kiran</surname></string-name>, <string-name><given-names>M. Y.</given-names> <surname>Javed</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Alhaisoni</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Tariq</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Nam</surname></string-name></person-group>, &#x201C;<article-title>Multi-layered deep learning features fusion for human action recognition</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>68</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>15</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Alhaisoni</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Armghan</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Alenezi</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Tariq</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Nam</surname></string-name></person-group>, &#x201C;<article-title>Video analytics framework for human action recognition</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>70</volume>, no. <issue>4</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>15</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. A.</given-names> <surname>Khan</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Hussain</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Xiaoming</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Yang</surname></string-name></person-group>, &#x201C;<article-title>An effective framework for driver fatigue recognition based on intelligent facial expressions analysis</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>6</volume>, no. <issue>5</issue>, pp. <fpage>67459</fpage>&#x2013;<lpage>67468</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Nawaz</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Nazir</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Javed</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Tariq</surname></string-name> and <string-name><given-names>M. A.</given-names> <surname>Khan</surname></string-name></person-group>, &#x201C;<article-title>An efficient deep learning approach to automatic glaucoma detection using optic disc and optic cup localization</article-title>,&#x201D; <source>Sensors</source>, vol. <volume>22</volume>, no. <issue>1</issue>, pp. <fpage>434</fpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Saleem</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Alhaisoni</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Tariq</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Armghan</surname></string-name> and <string-name><given-names>F.</given-names> <surname>Alenezi</surname></string-name></person-group>, &#x201C;<article-title>Human gait recognition: A single stream optimal deep learning features fusion</article-title>,&#x201D; <source>Sensors</source>, vol. <volume>21</volume>, no. <issue>5</issue>, pp. <fpage>7584</fpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Z. U.</given-names> <surname>Rehman</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Ahmed</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Dama&#x0161;evi&#x010D;ius</surname></string-name>, <string-name><given-names>S. R.</given-names> <surname>Naqvi</surname></string-name> and <string-name><given-names>W.</given-names> <surname>Nisar</surname></string-name></person-group>, &#x201C;<article-title>Recognizing apple leaf diseases using a novel parallel real-time processing framework based on MASK RCNN and transfer learning: An application for smart agriculture</article-title>,&#x201D; <source>IET Image Processing</source>, vol. <volume>15</volume>, no. <issue>9</issue>, pp. <fpage>2157</fpage>&#x2013;<lpage>2168</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Rashid</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Alhaisoni</surname></string-name>, <string-name><given-names>S. H.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>S. R.</given-names> <surname>Naqvi</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Rehman</surname></string-name></person-group>, &#x201C;<article-title>A sustainable deep learning framework for object recognition using multi-layers deep features fusion and selection</article-title>,&#x201D; <source>Sustainability</source>, vol. <volume>12</volume>, no. <issue>4</issue>, pp. <fpage>5037</fpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Koohzadi</surname></string-name> and <string-name><given-names>N. M.</given-names> <surname>Charkari</surname></string-name></person-group>, &#x201C;<article-title>Survey on deep learning methods in human action recognition</article-title>,&#x201D; <source>IET Computer Vision</source>, vol. <volume>11</volume>, no. <issue>2</issue>, pp. <fpage>623</fpage>&#x2013;<lpage>632</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Zahid</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Azam</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Sharif</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Kadry</surname></string-name> and <string-name><given-names>J. R.</given-names> <surname>Mohanty</surname></string-name></person-group>, &#x201C;<article-title>Pedestrian identification using motion-controlled deep neural network in real-time visual surveillance</article-title>,&#x201D; <source>Soft Computing</source>, vol. <volume>6</volume>, no. <issue>2</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>17</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Javed</surname></string-name>, <string-name><given-names>S. A.</given-names> <surname>Khan</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Saba</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Habib</surname></string-name> and <string-name><given-names>J. A.</given-names> <surname>Khan</surname></string-name></person-group>, &#x201C;<article-title>Human action recognition using fusion of multiview and deep features: An application to video surveillance</article-title>,&#x201D; <source>Multimedia Tools and Applications</source>, vol. <volume>13</volume>, no. <issue>2</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>27</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Krizhevsky</surname></string-name>, <string-name><given-names>I.</given-names> <surname>Sutskever</surname></string-name> and <string-name><given-names>G. E.</given-names> <surname>Hinton</surname></string-name></person-group>, &#x201C;<article-title>Imagenet classification with deep convolutional neural networks</article-title>,&#x201D; <source>Advances in Neural Information Processing Systems</source>, vol. <volume>25</volume>, no. <issue>5</issue>, pp. <fpage>1097</fpage>&#x2013;<lpage>1105</lpage>, <year>2012</year>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Szegedy</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ioffe</surname></string-name>, <string-name><given-names>V.</given-names> <surname>Vanhoucke</surname></string-name> and <string-name><given-names>A. A.</given-names> <surname>Alemi</surname></string-name></person-group>, &#x201C;<article-title>Inception-v4, inception-resnet and the impact of residual connections on learning</article-title>,&#x201D; in <conf-name>Thirty-first AAAI Conf. on Artificial Intelligence</conf-name>, <publisher-loc>NY, USA</publisher-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>6</lpage>, <year>2017</year>. </mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Akram</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Sharif</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Muhammad</surname></string-name>, <string-name><given-names>M. Y.</given-names> <surname>Javed</surname></string-name> and <string-name><given-names>S. R.</given-names> <surname>Naqvi</surname></string-name></person-group>, &#x201C;<article-title>Improved strategy for human action recognition; Experiencing a cascaded design</article-title>,&#x201D; <source>IET Image Processing</source>, vol. <volume>14</volume>, no. <issue>11</issue>, pp. <fpage>818</fpage>&#x2013;<lpage>829</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Alhaisoni</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Tariq</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Hussain</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Majid</surname></string-name> and <string-name><given-names>R.</given-names> <surname>Dama&#x0161;evi&#x010D;ius</surname></string-name></person-group>, &#x201C;<article-title>COVID-19 case recognition from chest CT images by deep learning, entropy-controlled firefly optimization, and parallel feature fusion</article-title>,&#x201D; <source>Sensors</source>, vol. <volume>21</volume>, no. <issue>2</issue>, pp. <fpage>7286</fpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Mittal</surname></string-name>, <string-name><given-names>L. M.</given-names> <surname>Goyal</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Roy</surname></string-name></person-group>, &#x201C;<article-title>A deep survey on supervised learning based human detection and activity classification methods</article-title>,&#x201D; <source>Multimedia Tools and Applications</source>, vol. <volume>4</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>57</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Bilal</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Maqsood</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Yasmin</surname></string-name>, <string-name><given-names>N. U.</given-names> <surname>Hasan</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Rho</surname></string-name></person-group>, &#x201C;<article-title>A transfer learning-based efficient spatiotemporal human action recognition framework for long and overlapping action classes</article-title>,&#x201D; <source>The Journal of Supercomputing</source>, vol. <volume>78</volume>, no. <issue>21</issue>, pp. <fpage>2873</fpage>&#x2013;<lpage>2908</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Sarkar</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Banerjee</surname></string-name>, <string-name><given-names>P. K.</given-names> <surname>Singh</surname></string-name> and <string-name><given-names>R.</given-names> <surname>Sarkar</surname></string-name></person-group>, &#x201C;<article-title>3D human action recognition: Through the eyes of researchers</article-title>,&#x201D; <source>Expert Systems with Applications</source>, vol. <volume>17</volume>, no. <issue>7</issue>, pp. <fpage>116424</fpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Ying</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Yang</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Hu</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Liu</surname></string-name></person-group>, &#x201C;<article-title>Improved human action recognition approach based on two-stream convolutional neural network model</article-title>,&#x201D; <source>The Visual Computer</source>, vol. <volume>37</volume>, no. <issue>2</issue>, pp. <fpage>1327</fpage>&#x2013;<lpage>1341</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>V. A.</given-names> <surname>Chenarlogh</surname></string-name> and <string-name><given-names>F.</given-names> <surname>Razzazi</surname></string-name></person-group>, &#x201C;<article-title>Multi-stream 3D CNN structure for human action recognition trained by limited data</article-title>,&#x201D; <source>IET Computer Vision</source>, vol. <volume>13</volume>, no. <issue>10</issue>, pp. <fpage>338</fpage>&#x2013;<lpage>344</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Sharif</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Javed</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Gulfam</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Iqbal</surname></string-name> and <string-name><given-names>T.</given-names> <surname>Saba</surname></string-name></person-group>, &#x201C;<article-title>Intelligent human action recognition: A framework of optimal features selection based on euclidean distance and strong correlation</article-title>,&#x201D; <source>Journal of Control Engineering and Applied Informatics</source>, vol. <volume>21</volume>, no. <issue>15</issue>, pp. <fpage>3</fpage>&#x2013;<lpage>11</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>N.</given-names> <surname>Jaouedi</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Boujnah</surname></string-name> and <string-name><given-names>M. S.</given-names> <surname>Bouhlel</surname></string-name></person-group>, &#x201C;<article-title>A new hybrid deep learning model for human action recognition</article-title>,&#x201D; <source>Journal of King Saud University-Computer and Information Sciences</source>, vol. <volume>32</volume>, no. <issue>20</issue>, pp. <fpage>447</fpage>&#x2013;<lpage>453</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Sharif</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Akram</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Raza</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Saba</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Rehman</surname></string-name></person-group>, &#x201C;<article-title>Hand-crafted and deep convolutional neural network features fusion and selection strategy: An application to intelligent human action recognition</article-title>,&#x201D; <source>Applied Soft Computing</source>, vol. <volume>87</volume>, no. <issue>21</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>26</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Abdelbaky</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Aly</surname></string-name></person-group>, &#x201C;<article-title>Human action recognition using three orthogonal planes with unsupervised deep convolutional neural network</article-title>,&#x201D; <source>Multimedia Tools and Applications</source>, vol. <volume>80</volume>, no. <issue>32</issue>, pp. <fpage>20019</fpage>&#x2013;<lpage>20043</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Afza</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Sharif</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Kadry</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Manogaran</surname></string-name> and <string-name><given-names>T.</given-names> <surname>Saba</surname></string-name></person-group>, &#x201C;<article-title>A framework of human action recognition using length control features fusion and weighted entropy-variances based feature selection</article-title>,&#x201D; <source>Image and Vision Computing</source>, vol. <volume>106</volume>, no. <issue>17</issue>, pp. <fpage>104090</fpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Abdelbaky</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Aly</surname></string-name></person-group>, &#x201C;<article-title>Human action recognition using short-time motion energy template images and PCANet features</article-title>,&#x201D; <source>Neural Computing and Applications</source>, vol. <volume>32</volume>, no. <issue>9</issue>, pp. <fpage>12561</fpage>&#x2013;<lpage>12574</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. P.</given-names> <surname>Sahoo</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ari</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Mahapatra</surname></string-name> and <string-name><given-names>S. P.</given-names> <surname>Mohanty</surname></string-name></person-group>, &#x201C;<article-title>HAR-Depth: A novel framework for human action recognition using sequential learning and depth estimated history images</article-title>,&#x201D; <source>IEEE Transactions on Emerging Topics in Computational Intelligence</source>, vol. <volume>21</volume>, no. <issue>11</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>13</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Muhammad</surname></string-name>, <string-name><given-names>U. Amin</given-names></string-name>, <string-name><given-names>A. S.</given-names> <surname>Imran</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Sajjad</surname></string-name></person-group>, &#x201C;<article-title>Human action recognition using attention based LSTM network with dilated CNN features</article-title>,&#x201D; <source>Future Generation Computer Systems</source>, vol. <volume>125</volume>, no. <issue>7</issue>, pp. <fpage>820</fpage>&#x2013;<lpage>830</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Sharif</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Akram</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Dama&#x0161;evi&#x010D;ius</surname></string-name> and <string-name><given-names>R.</given-names> <surname>Maskeli&#x016B;nas</surname></string-name></person-group>, &#x201C;<article-title>Skin lesion segmentation and multiclass classification using deep learning features and improved moth flame optimization</article-title>,&#x201D; <source>Diagnostics</source>, vol. <volume>11</volume>, no. <issue>1</issue>, pp. <fpage>811</fpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Yadav</surname></string-name>, <string-name><given-names>J. K.</given-names> <surname>Sandhu</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Pathak</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Jadhav</surname></string-name></person-group>, &#x201C;<article-title>Chest x-ray scanning based detection of COVID-19 using deepconvolutional neural network</article-title>,&#x201D; <source>Future Generation Computer Systems</source>, vol. <volume>125</volume>, no. <issue>7</issue>, pp. <fpage>820</fpage>&#x2013;<lpage>830</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Tabernik</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Kristan</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Leonardis</surname></string-name></person-group>, &#x201C;<article-title>Spatially-adaptive filter units for compact and efficient deep neural networks</article-title>,&#x201D; <source>International Journal of Computer Vision</source>, vol. <volume>128</volume>, no. <issue>61</issue>, pp. <fpage>2049</fpage>&#x2013;<lpage>2067</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Mirjalili</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Lewis</surname></string-name></person-group>, &#x201C;<article-title>The whale optimization algorithm</article-title>,&#x201D; <source>Advances in Engineering Software</source>, vol. <volume>95</volume>, no. <issue>17</issue>, pp. <fpage>51</fpage>&#x2013;<lpage>67</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>B. S.</given-names> <surname>Kumar</surname></string-name>, <string-name><given-names>S. V.</given-names> <surname>Raju</surname></string-name> and <string-name><given-names>H. V.</given-names> <surname>Reddy</surname></string-name></person-group>, &#x201C;<article-title>Human action recognition using a novel deep learning approach</article-title>,&#x201D; <source>Materials Science and Engineering</source>, vol. <volume>1042</volume>, no. <issue>31</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>17</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Kiran</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Younus Javed</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Alhaisoni</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Tariq</surname></string-name> and <string-name><given-names>Y.</given-names> <surname>Nam</surname></string-name></person-group>, &#x201C;<article-title>Multi-layered deep learning features fusion for human action recognition</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>69</volume>, no. <issue>3</issue>, pp. <fpage>4061</fpage>&#x2013;<lpage>4075</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Ahmed</surname></string-name>, <string-name><given-names>M. H.</given-names> <surname>Yousaf</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Yasin</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Maqsood</surname></string-name></person-group>, &#x201C;<article-title>Robust suspicious action recognition approach using pose descriptor</article-title>,&#x201D; <source>Mathematical Problems in Engineering</source>, vol. <volume>2021</volume>, no. <issue>12</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>12</lpage>, <year>2021</year>.</mixed-citation></ref>
</ref-list>
</back>
</article>
















