<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">27571</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2022.027571</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Automatic Detection of Weapons in Surveillance Cameras Using Efficient-Net</article-title>
<alt-title alt-title-type="left-running-head">Automatic Detection of Weapons in Surveillance Cameras Using Efficient-Net</alt-title>
<alt-title alt-title-type="right-running-head">Automatic Detection of Weapons in Surveillance Cameras Using Efficient-Net</alt-title>
</title-group>
<contrib-group content-type="authors">
<contrib id="author-1" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Arif</surname><given-names>Erssa</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><email>erssaarif1@gmail.com</email>
</contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Shahzad</surname><given-names>Syed Khuram</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Iqbal</surname><given-names>Muhammad Waseem</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Jaffar</surname><given-names>Muhammad Arfan</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Alshahrani</surname><given-names>Abdullah S.</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Alghamdi</surname><given-names>Ahmed</given-names></name><xref ref-type="aff" rid="aff-6">6</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Department of Computer Science, Superior University</institution>, <addr-line>Lahore, 54000</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-2"><label>2</label><institution>Department of Informatics &#x0026; Systems, University of Management &#x0026; Technology</institution>, <addr-line>Lahore, 54000</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-3"><label>3</label><institution>Department of Software Engineering, Superior University</institution>, <addr-line>Lahore, 54000</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-4"><label>4</label><institution>Faculty of Computer Science &#x0026; Information Technology, Superior University</institution>, <addr-line>Lahore, 54000</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-5"><label>5</label><institution>Department of Computer Science &#x0026; Artificial Intelligence, College of Computer Science &#x0026; Engineering, University of Jeddah</institution>, <addr-line>Jeddah, 21493</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-6"><label>6</label><institution>Department of Software Engineering, College of Computer Science and Engineering, University of Jeddah</institution>, <addr-line>Jeddah, 21493</addr-line>, <country>Saudi Arabia</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Erssa Arif. Email: <email>erssaarif1@gmail.com</email></corresp>
</author-notes>
<pub-date pub-type="epub" date-type="pub" iso-8601-date="2022-04-20"><day>20</day>
<month>04</month>
<year>2022</year></pub-date>
<volume>72</volume>
<issue>3</issue>
<fpage>4615</fpage>
<lpage>4630</lpage>
<history>
<date date-type="received"><day>20</day><month>1</month><year>2022</year></date>
<date date-type="accepted"><day>08</day><month>3</month><year>2022</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2022 Arif et al.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Arif et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_27571.pdf"></self-uri>
<abstract>
<p>The conventional Close circuit television (CCTV) cameras-based surveillance and control systems require human resource supervision. Almost all the criminal activities take place using weapons mostly a handheld gun, revolver, pistol, swords etc. Therefore, automatic weapons detection is a vital requirement now a day. The current research is concerned about the real-time detection of weapons for the surveillance cameras with an implementation of weapon detection using Efficient&#x2013;Net. Real time datasets, from local surveillance department&#x0027;s test sessions are used for model training and testing. Datasets consist of local environment images and videos from different type and resolution cameras that minimize the idealism. This research also contributes in the making of Efficient-Net that is experimented and results in a positive dimension. The results are also been represented in graphs and in calculations for the representation of results during training and results after training are also shown to represent our research contribution. Efficient-Net algorithm gives better results than existing algorithms. By using Efficient-Net algorithms the accuracy achieved 98.12&#x0025; when epochs increase as compared to other algorithms.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Detection algorithms</kwd>
<kwd>machine learning</kwd>
<kwd>machine vision</kwd>
<kwd>video surveillance</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1"><label>1</label><title>Introduction</title>
<p>Today&#x0027;s, a huge number of criminal activities are taking place using handheld arms e.g., guns, pistols, revolver and semi machine guns of shot guns also in some cases [<xref ref-type="bibr" rid="ref-1">1</xref>]. These activities can be reduced by monitoring and identifying at early stage. The way to minimize the violence is by the early detection of suspicious activity so that the law enforcements can take necessary action. Current control and surveillance still need human supervision and interference. Close circuit television (CCTV) surveillance cameras are broadly in use for monitoring and the other security purposes [<xref ref-type="bibr" rid="ref-2">2</xref>]. Currently, deep learning approaches are increasingly adopted because of the capability of giving data-driven solutions to such problems [<xref ref-type="bibr" rid="ref-3">3</xref>]. Extraordinary results of image classification using deep neural networks have exceeded the human performance [<xref ref-type="bibr" rid="ref-4">4</xref>]. <xref ref-type="fig" rid="fig-1">Fig. 1</xref> illustrates an abstraction of common process flow of such detection applications.</p>
<fig id="fig-1"><label>Figure 1</label><caption><title>Detection and tracking application process flow</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-1.png"/></fig>
<sec id="s1_1"><label>1.1</label><title>Data Detection and Extraction</title>
<p>The first step is data extraction and detection model training using deep learning-based algorithms. There are varieties of different algorithms used for object detection. Histogram of oriented gradients (HOG) is used as a feature descriptor for the detection of the objects. It is comprised of gradient occurrences and their orientations in a localized part of image like Region of interest (ROI) and detection window etc. The basic advantage of HOG is that it is easy to implement and understand&#x00A0;[<xref ref-type="bibr" rid="ref-5">5</xref>].</p>
<p>Single shot detector (SSD) is a single deep network to detect objects from an image. This algorithm is easy to integrate into multiple systems that require the detection of objects [<xref ref-type="bibr" rid="ref-6">6</xref>]. The Spatial pyramid pooling) (SPP-net) has the ability to generate a representation of fixed length irrespective of the image sixe or scale. Pyramid pooling shown robust results for object deformation and the performance of all Convolutional based neural networks can be improved by using SPP-net [<xref ref-type="bibr" rid="ref-7">7</xref>].</p>
</sec>
<sec id="s1_2"><label>1.2</label><title>Efficient-Net Architecture</title>
<p>Efficient-Net formulates the basic backbone of Efficient-Net by studying the scaling method of ConvNet systems. It is quite tedious for researchers to add more layers to ConvNet, make them wider and deeper with high resolutions. To tackle all these problems, Efficient-Net model is used by using a few numbers of Floating point operation per second (FLOPS) to create a baseline ConvNet, which is called Efficient-Net B0. After that, Efficient-Net B1 is built by scaling up Efficient-Net B0. This scaling function is then applied to Efficient-Net B7. This Efficient-Net architecture is further used to form Efficient-Net, which used fused features with different resolutions for object detection [<xref ref-type="bibr" rid="ref-8">8</xref>].</p>
<p>All the above-mentioned techniques have already been used by researchers for the detection of concealed weapons. These algorithms also help in detecting a weapon under a loose clothing, its shape, size, type of weapon and total number of weapons. Each technique has certain benefits for various scenarios in terms of speed, efficiency and accuracy. By using Efficient-Net algorithm, the accuracy increase as compared to other algorithms. Like CNN, RCNN etc [<xref ref-type="bibr" rid="ref-9">9</xref>].</p>
<p>The purpose of this work is to dive into existing systems and conduct brief research and proposes a solution that overcomes the drawback on some new and state of the art techniques or methodologies, on real-time gathered data in deep learning. Hence, the aim of this work is to represent automated hand-held weapon detection and alarm system by using Convolutional neural network (CNN) deep learning algorithm.</p>
<p>Detecting a gun is a challenge because of its many subtleties. Using CNNs for automatic detection of guns in video arise several challenges:
<list list-type="bullet">
<list-item><p>Weapon may be handled with tow or one hand in several ways so the large part of the weapon is occluded.</p></list-item>
<list-item><p>Designing a new data set is time taking.</p></list-item>
<list-item><p>Automatic Weapon detection triggering alarm in real-time.</p></list-item>
<list-item><p>Accurate location of weapon in the scene for triggering alarm.</p></list-item>
</list></p>
<p>Some other problems can also occur while detecting a weapon and are necessity of real-time processing such as deformation of gun and noise in an image [<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-6">6</xref>,<xref ref-type="bibr" rid="ref-8">8</xref>].</p>
</sec>
</sec>
<sec id="s2"><label>2</label><title>Literature Review</title>
<p>In this field, most of the previous research depends upon either context or pose and does not allow our system to deal with both factors. Using this approach, they built a model based on You only looks once (YOLO) neural network, which is able to classify both kind of (low level and high level) threats with an efficiency of 84&#x0025;.</p>
<p>Multiple simulated experiment to classify and evaluate disease found in five cassava leaf data set and our framework is capable of producing relatively accurate classification results despite small difference between the test set and image [<xref ref-type="bibr" rid="ref-10">10</xref>].</p>
<p>To achieve a high detection rate, the authors of this research increased the total number of images by taking different possible directions of pistols [<xref ref-type="bibr" rid="ref-8">8</xref>]. The functionality of the architectures of Convolutional neural networking (CNN) by giving the complete description of CNN models, which started from LeNet model and further involved ZFNet, VGGNet, AlexNet, SENet, ResNet, GoogleNet, ResNeXt, Xception, PNAS/ENAS and DenseNet [<xref ref-type="bibr" rid="ref-9">9</xref>].</p>
<p>The experiment results show the improved model is more efficient and finally achieved the optimal accuracy of 99.69&#x0025;. Compared with other human behavior recognition methods has stronger environment and lower human invasivene [<xref ref-type="bibr" rid="ref-10">10</xref>].</p>
<p>Deep convolutional network (DCN), a revolutionary model: F-R CNN model, through transfer learning and evaluated the weapon detection on IMFDB which is a standard weapon database. In the paper CNN implementation using MatConvNet, MATLAB toolbox for the implementation of Convolutional neural networks for the applications of computer vision without the usage of Graphical processing unit (GPU) [<xref ref-type="bibr" rid="ref-10">10</xref>]. Each CCTV image was able to capture the image by taking care of indoor and outdoor conditions with different resolutions to represent various scales of gun. To train data, M2Det network was used and then this trained network was authenticated by taking images from the dataset of University of central florida (UCF) crime videos. The experimental results of this research indicated that by using the proposed model, the average accuracy of weapon detection can be increased up to 18&#x0025; when we compared it with the previous approaches [<xref ref-type="bibr" rid="ref-11">11</xref>].</p>
<p>To improve the results of object detection and its classification the domain of terrorism and military, a Multi spectral fusion system (MSFMT) is presented in this paper. This system mainly depends on the combination of dempster-shafer statistical method and deep learning techniques. In this research, MSFMT system is used to help in improving the results of classifications by creating an algorithm for fusion between multiple spectrums [<xref ref-type="bibr" rid="ref-12">12</xref>].</p>
<p>Single shot detection (SSD) and Region convolutional neural network (RCNN) for self-created and pre labeled image dataset for the detection of weapon. The experimental results showed that both were efficient algorithms but their real time application gave results based on a compromise between accuracy and speed. Faster RCNN method was better in terms of accuracy as it gave the accuracy of 84.6&#x0025; and the accuracy of SSD method was 73.8&#x0025; [<xref ref-type="bibr" rid="ref-13">13</xref>].</p>
<p>CCTV depends on human supervision that may cause human prone errors such as a person can miss some crime events while monitoring multiple screens at the same time. To tackle this situation, a crime intension detection system to detect crimes happening in real time images, videos and after detection this system sends warnings to human supervisor by using Short message service (SMS) sending module. Fast RCNN and RCNN methods were also used to mark the objects in the CCTV images like knife, gun, pistol and person [<xref ref-type="bibr" rid="ref-14">14</xref>].</p>
<p>Deep learning-based algorithms were used for object classification and detection. By using techniques of sensor fusion, a framework consisting of multi-sensor data was not only designed but also embedded by extracting the features of image modules using Raspberry pi and intel movidious stick. This framework helped in reducing the sub problems like resolution, noise by the implementation of a modified R-CNN algorithm [<xref ref-type="bibr" rid="ref-15">15</xref>]. Not only the techniques of recognizing the features of a moving object are explained in this research work but also its explained that how to classify the concealed objects present in the video frames. This paper also reviews some research gaps like it is difficult to identify the concealed objects in loose cloths and the shape and size of weapon varies, etc [<xref ref-type="bibr" rid="ref-16">16</xref>].</p>
<p>CNN framework is used to classify handguns in the CCTV video frames by only using edge features. Moreover, IAGMM and ViBe algorithms are used to evaluate the experimental models. These algorithms are important to give more inner detail and they have high capacity of resisting sudden changes and noise, which could happen while operating in an outdoor environment. They simulate the results by taking 1869 positive and 4000 negative images to train the CNN model [<xref ref-type="bibr" rid="ref-17">17</xref>]. Object detection structures based on deep learning are reviewed to solve many sub problems like low resolution, clutter and blocking by making several modifications in the R-CNN method. This research also provides the experimental analyses to make a comparison between different methods like R-CNN, YOLO and CNN. This review gives a basic architecture for object detection, pedestrian detection and facial detection. Finally, they proposed to work on multimodal information fusion, multitask joint optimization, contextual modeling, spatial correlations and scale adaption [<xref ref-type="bibr" rid="ref-14">14</xref>,<xref ref-type="bibr" rid="ref-16">16</xref>].</p>
<p>By carrying out a detailed study of the previous state-of-the-are detection models, we were able to enhance the computational efficiency which led to use EfficientDet model. EfficientDet as a model needs less Floating point operations per second (FLOPS) than YOLO, also its the needs much less parameters compared to algorithms like RCNN, Mask R-CNN, etc. which reduces the complexity and training time of the model. Most of the prior object detection algorithm use top to bottom or either bottom to top approach, while using regular efficient connections; EfficientDet allows the flow of information in both directions using BiFPN [<xref ref-type="bibr" rid="ref-18">18</xref>].</p>
</sec>
<sec id="s3"><label>3</label><title>Methodology</title>
<p>The research methodology comprised of two basic steps first the research planning and the second is research execution. The research plan comprised of two tasks including research problem identification and consequent research experiments directions while the second task is data acquisition for the model training and testing. The research execution involves a step-by-step algorithm application at the collected data sets for training and testing for the model generated using Efficient-Net.</p>
<sec id="s3_1"><label>3.1</label><title>Research Plan</title>
<sec id="s3_1_1"><label>3.1.1</label><title>Problem Statement</title>
<p>Main aim for using multi-scale feature fusion is to aggregating features in many resolutions. It gives us a list of different multi scale features as in which is representation of feature at particular level but, objective of using multi scale feature fusion is to locate the transformation that could efficiently combine the different features to generate new features list.</p>
</sec>
<sec id="s3_1_2"><label>3.1.2</label><title>Data Collection</title>
<p>We collect data from session training data in our research like public data and use data with the permission of the authority. The recorded live stream video from the CCTV camera is first preprocessed into frames to clean data and avoid any noise. Later these frames are labelled by annotation to train the model. We apply algorithm on labelled images extracted from videos.</p>
</sec>
</sec>
<sec id="s3_2"><label>3.2</label><title>Research Execution</title>
<p>The step-by-step research execution flow is illustrated in the <xref ref-type="fig" rid="fig-2">Fig. 2</xref>.</p>
<fig id="fig-2"><label>Figure 2</label><caption><title>Research execution flow</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-2.png"/></fig>
<sec id="s3_2_1"><label>3.2.1</label><title>Cross Scale Connection (CSC)</title>
<p>Conventional top-down Feature pyramid network (FPN) is fundamentally constrained by the one-way flow of information. To tackle this situation, Path of network (PANet) architecture contains an additional network, which follows the bottom-up path. Moreover, for finding more suitable feature network topology, neural architecture is used by Neural Architecture Search (NAS_FPN). But its drawback is that it requires much GPU consumption during search and beside this it is difficult to modify/interpret the found network.</p>
<p>After analyzing efficiency and performance of above- mentioned networks, it is observed that PANet able to achieve maximum accuracy as compare to others. Nevertheless, PANet uses a greater number of parameters and flop in it so required more Computation power. For achievement of more accuracy of the model, we proposed a methodology in which our main aim is to reducing the parameters and to achieve the better accuracy with a lesser number of modes. At first, after inspection of architecture shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, we excluded those layers which have only single input [<xref ref-type="bibr" rid="ref-9">9</xref>].</p>
<fig id="fig-3"><label>Figure 3</label><caption><title>Directional path of networks</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-3.png"/></fig>
<p>Intuition behind this is as simple that if a layer only has one input edge it means its contribution is minimal in network for fusing different features. By doing this, we are able to achieve simplified bidirectional architecture. In the second step, when they (input and output node) are at the same point, we add one extra edge from the original node to the destination node to fuse additional features with less cost. At last, for enabling a larger number of high-level feature fusions, we use every bidirectional path as a single layer for extracting features and utilize it multiple times. In this way, we differentiate our network from PANet that contains only one bidirectional path, we differentiate our network from PANet that contains only one bidirectional path&#x00A0;[<xref ref-type="bibr" rid="ref-19">19</xref>].</p>
<p>After analyzing efficiency and performance of above- mentioned networks, it is observed that PANet able to achieve maximum accuracy as compare to others. Nevertheless, PANet uses a greater number of parameters and flop in it so required more Computation power. For achievement of more accuracy of the model, we proposed a methodology in which our main aim is to reducing the parameters and to achieve the better accuracy with a lesser number of modes. At first, after inspection of architecture we excluded those layers which have only single input. Intuition behind this is as simple: that if a layer only has one input edge it means its contribution is minimal in network for fusing different features [<xref ref-type="bibr" rid="ref-13">13</xref>].</p>
<p>By doing this, we are able to achieve simplified bidirectional architecture. In the second step, when they (input and output node) are at the same point, we add one extra edge from the original node to the destination node to fuse additional features with less cost. At last, for enabling a larger number of high-level feature fusions, we use every bidirectional path as a single layer for extracting features and utilize it multiple times. In this way, we differentiate our network from PANet that contains only one bidirectional path, we differentiate our network from PANet that contains only one bidirectional path&#x00A0;[<xref ref-type="bibr" rid="ref-19">19</xref>].</p>
</sec>
<sec id="s3_2_2"><label>3.2.2</label><title>Fusion of Weighted Features</title>
<p>A common method of fusing feature with different resolutions is to first resize them to the same resolution and then summarize them. The traditional method uses the approach of considering all features with equal contribution. However, we notice that because specific input characteristics are at various resolutions, they typically contribute unequally to the output function. We suggest introducing an additional weight for each input to resolve this problem, and making the network know the value of each input feature. According to this, we used three different weighted fusion techniques [<xref ref-type="bibr" rid="ref-20">20</xref>].</p>
</sec>
<sec id="s3_2_3"><label>3.2.3</label><title> Unbounded Fusion: O &#x003D; &#x03A3;I WI. II</title>
<p>Here WI indicating a weight to learn and which can be treated as a multidimensional tensor i.e., per pixel. This learnable weight can also be treated as a scaler and as a vector according to need. It&#x0027;s also observed that it is possible to achieve comparable accuracy with less computation in comparison with other different approaches. Nevertheless, because the scalar weight is unbounded, this may theoretically contribute to uncertainty in preparation. Consequently, we turn to weight normalization to restrict the significance spectrum of increasing weight&#x00A0;[<xref ref-type="bibr" rid="ref-19">19</xref>].</p>
</sec>
<sec id="s3_2_4"><label>3.2.4</label><title>Softmax-Based Fusion</title>
<p>An interesting concept is to apply softmax to each weight, to normalize all weights to be likelihood with a significance set of 0 To 1, which reflects the value of each data. But, it leads us to extra latency cost on GPU. To tackle this problem, a fast fusion approach is proposed by us [<xref ref-type="bibr" rid="ref-20">20</xref>].</p>
</sec>
<sec id="s3_2_5"><label>3.2.5</label><title>Fast Normalized Fusion</title>
<p>In this to track the feature whose weight is&#x2009;&#x003E;&#x2009;0 we used Relu to avoid any numerical instability. Likewise, the value of increasing weighted weight still dropped between 0 and 1, but it is much more effective because there is no softmax process here. This make sure that the fast fusion method produces same learning accuracy as softmax but its computation on GPUs is 30 percent less than as compare to softmax. We named this network as a Bidirectional FPN. Our final model, combine the fast-normalized fusion and bidirectional cross scale network.
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:msubsup><mml:mi>P</mml:mi><mml:mn>6</mml:mn><mml:mrow><mml:mi>t</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>.</mml:mo><mml:mspace width="thickmathspace" /><mml:msubsup><mml:mi>P</mml:mi><mml:mn>6</mml:mn><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow><mml:mspace width="thickmathspace" /><mml:mo>.</mml:mo><mml:mspace width="thickmathspace" /><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>z</mml:mi><mml:mi>e</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>P</mml:mi><mml:mn>7</mml:mn><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03B5;</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:msubsup><mml:mi>P</mml:mi><mml:mn>6</mml:mn><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>v</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>w</mml:mi><mml:mrow><mml:msubsup><mml:mi></mml:mi><mml:mn>1</mml:mn><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>.</mml:mo><mml:mspace width="thickmathspace" /><mml:msubsup><mml:mi>P</mml:mi><mml:mn>6</mml:mn><mml:mrow><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:mi>w</mml:mi><mml:mrow><mml:msubsup><mml:mi></mml:mi><mml:mn>2</mml:mn><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>.</mml:mo><mml:mspace width="thickmathspace" /><mml:msubsup><mml:mi>P</mml:mi><mml:mn>6</mml:mn><mml:mrow><mml:mi>t</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:mi>w</mml:mi><mml:mrow><mml:msubsup><mml:mi></mml:mi><mml:mn>3</mml:mn><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>.</mml:mo><mml:mspace width="thickmathspace" /><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>z</mml:mi><mml:mi>e</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi>P</mml:mi><mml:mn>6</mml:mn><mml:mrow><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mn>3</mml:mn></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03B5;</mml:mi></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>Ptd is showing the intermediate features of level 6. Moreover, the output features of level 6 are indicated by and these are on bottom to up route shown in in <xref ref-type="disp-formula" rid="eqn-1">Eqs. (1)</xref> and <xref ref-type="disp-formula" rid="eqn-2">(2)</xref>. Construction of other features is carried in the same manner. We also use the depth wise separable technique in order to increase the efficiency for feature fusion. After each convolution we are adding batch for activation and normalization [<xref ref-type="bibr" rid="ref-21">21</xref>].</p>
</sec>
<sec id="s3_2_6"><label>3.2.6</label><title>Detection Architecture</title>
<p>On the basis of our bidirectional FPN, we proposed a new method of detection of object with lesser computation and parameters. This section includes the discussion of our network&#x0027;s architecture and how we proposed a new method of compound scaling for our model.</p>
<p><xref ref-type="fig" rid="fig-4">Fig. 4</xref> shows the architectural diagram of our proposed model. In which it is clearly visible that our model is utilizing one stage detector paradigm. As the back bone of our network we used efficient nets which are pre-trained ImageNets.</p>
<fig id="fig-4"><label>Figure 4</label><caption><title>Level of features of efficient backbone network</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-4.png"/></fig>
<p>Bidirectional FPN proposed by us is serve as the network which is responsible for feature extraction. Which utilize feature from level 3&#x2013;7 from efficient net and then in loop applies the technique of bottom-up and top down two-way feature fusion (as shown in <xref ref-type="table" rid="table-1">Tab. 1</xref> below).</p>
<table-wrap id="table-1"><label>Table 1</label><caption><title>Level of features of network</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Symbol</th>
<th align="left">Input size</th>
<th align="left">Backbone network</th>
<th align="left">BiFPN channel</th>
<th align="left">Box layers</th>
<th align="left">Class layers</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">D0(&#x0278;&#x003D;0)</td>
<td align="left">512</td>
<td align="left">B0</td>
<td align="left">64</td>
<td align="left">3</td>
<td align="left">3</td>
</tr>
<tr>
<td align="left">D1(&#x0278;&#x003D;1)</td>
<td align="left">640</td>
<td align="left">B1</td>
<td align="left">88</td>
<td align="left">4</td>
<td align="left">3</td>
</tr>
<tr>
<td align="left">D2(&#x0278;&#x003D;2)</td>
<td align="left">768</td>
<td align="left">B2</td>
<td align="left">112</td>
<td align="left">5</td>
<td align="left">3</td>
</tr>
<tr>
<td align="left">D3(&#x0278;&#x003D;3)</td>
<td align="left">896</td>
<td align="left">B3</td>
<td align="left">160</td>
<td align="left">6</td>
<td align="left">4</td>
</tr>
<tr>
<td align="left">D4(&#x0278;&#x003D;4)</td>
<td align="left">1024</td>
<td align="left">B4</td>
<td align="left">224</td>
<td align="left">7</td>
<td align="left">4</td>
</tr>
<tr>
<td align="left">D5(&#x0278;&#x003D;5)</td>
<td align="left">1280</td>
<td align="left">B5</td>
<td align="left">288</td>
<td align="left">7</td>
<td align="left">4</td>
</tr>
<tr>
<td align="left">D6(&#x0278;&#x003D;6)</td>
<td align="left">1280</td>
<td align="left">B6</td>
<td align="left">384</td>
<td align="left">8</td>
<td align="left">5</td>
</tr>
<tr>
<td align="left">D7</td>
<td align="left">1536</td>
<td align="left">B6</td>
<td align="left">384</td>
<td align="left">8</td>
<td align="left">5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Such fused features are fed to a class and box network respectively to generate predictions of object type and bounding boxes. Moreover, weights of box-network and class are used in all level of features [<xref ref-type="bibr" rid="ref-17">17</xref>,<xref ref-type="bibr" rid="ref-19">19</xref>].</p>
</sec>
<sec id="s3_2_7"><label>3.2.7</label><title>Compound-Scaling</title>
<p>To maximize both accuracy and performance, we would like to build a collection of models capable of meeting a broad range of resource constraints. One main question here is how to expand the baseline of our proposed model. Earlier methodology mostly used deeper networks as backbone to scale up their models i.e., ResNet, Amoeba Net [<xref ref-type="bibr" rid="ref-7">7</xref>].</p>
<p>These models require the large number of layers and huge number of FLOPs. Moreover, one of the drawbacks of these networks is they support limited scaling dimensions. However, in the recent timeline a number of networks show comparable performance and effective results on classification of image. As they carried out in depth analysis of features by enhancing all dimensions of the proposed networks (width, depth, height). By studying those networks, we proposed a novel scaling method i.e., &#x2018;Compound Scaling&#x2019; for detection of different objects. The proposed technique of scaling utilizes &#x03C6; (which is simple compound coefficient) to enhance all the dimensions of backbone, bidirectional FPN, resolution and box/class network together [<xref ref-type="bibr" rid="ref-21">21</xref>,<xref ref-type="bibr" rid="ref-22">22</xref>].</p>
<p>For scaling we used heuristic based technique in order to prevent the large number of scaling dimensions. Because an object detector has a large number of scaling dimensions as compared to image classification. But we keep using our ideas of scaling up all the dimensions jointly.</p>
</sec>
<sec id="s3_2_8"><label>3.2.8</label><title>Backbone-Network</title>
<p>To use the pertained checkpoints of ImageNet easily we employ the efficient-net with the same coefficients of width/depth.</p>
</sec>
<sec id="s3_2_9"><label>3.2.9</label><title>H. Class/Box Prediction Network</title>
<p>To use pertained checkpoints of ImageNet easily we employ the efficient-net with the same coefficients of width/depth. To make the width same as bidirectional FPN and to increase the depth linearly we use the equation [<xref ref-type="bibr" rid="ref-19">19</xref>,<xref ref-type="bibr" rid="ref-23">23</xref>].</p>
<p>Use the following equation Presented in <xref ref-type="disp-formula" rid="eqn-3">Eq. (3)</xref>.
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>o</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mn>3</mml:mn><mml:mo>+</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:mstyle displaystyle="false" scriptlevel="0"><mml:mrow><mml:mrow><mml:mspace width="thickmathspace" /><mml:mi mathvariant="normal">&#x2205;</mml:mi></mml:mrow></mml:mrow></mml:mstyle><mml:mrow><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true">/</mml:mo><mml:mrow><mml:mrow><mml:mrow><mml:mspace width="thickmathspace" /><mml:mi mathvariant="normal">&#x2205;</mml:mi></mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:mrow><mml:mstyle displaystyle="false" scriptlevel="0"><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:mstyle></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>In <xref ref-type="fig" rid="fig-5">Fig. 5</xref> we clearly show that this technique of scaling significantly enhances the efficiency as compared to single dimension scaling method.</p>
<fig id="fig-5"><label>Figure 5</label><caption><title>Technique of scaling</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-5.png"/></fig>
<p>In <xref ref-type="fig" rid="fig-5">Fig. 5</xref> shown different ratio of FLOPS and COCOAP with all techniques in different color and also shown enhance the efficiency and show the value FLOPS with different color. The compound scale COCO AP is round about 46 and scale by image size COCO AP is 38 etc.</p>
</sec>
</sec>
</sec>
<sec id="s4"><label>4</label><title>Implementation Details</title>
<p>The architecture of the proposed network is designed in Tensorflow framework. The training and the validation process is performed using Tensorflow. The framework is installed on Ubuntu 16.04 with python 3.5 language. The hardware and software specification are as follow.</p>
<sec id="s4_1"><label>4.1</label><title>Hardware Specification</title>
<p>The training of model is performed on Corei7 (7th generation) CPU has 8 cores with 32 GB DDR3 RAM. NVIDIA GPU 1080 Ti having 11GB DDR5 memory and 3584 CUDA cores are used for parallel processing and matrix multiplication and other math operations. We used the mini batch of 200 samples while training and the validation. The Mean square error (MSE) is used for loss calculation. The mathematical explanation of MSE is described in <xref ref-type="disp-formula" rid="eqn-4">Eq. (4)</xref>. SGD is used for optimization of weights along the dynamic learning rate. The dynamic learning rate along the epochs is shown in <xref ref-type="table" rid="table-1">Tab. 1</xref> [<xref ref-type="bibr" rid="ref-24">24</xref>].
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mi>d</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mi>q</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo>+</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>+</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mo>+</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>+</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mi>q</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow><mml:mspace width="thickmathspace" /></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:msqrt></mml:math></disp-formula></p>
<p>Here p is predicted and q actual labels of the annotated dataset provided while training and validation, p1 represents the result of the 1st sample predicted by the trained model and the q1 represents the ground truth of 1st sample annotated by the human.</p>
<p>The above table shows the dynamic learning rate that how much learning rate will decrease after each epoch in fractions between a specified range of the epochs. The model is train for 1000 epochs with mini batch of 200. The learning rate from epochs 1 to 500 is 0.001, from 501 to 800 is 0.0001 and from 801 to 1000 is 0.00001 as shown in <xref ref-type="table" rid="table-2">Tab. 2</xref>. The reason of dynamic learning rate is to accelerate the training process in initial steps. The higher learning rate means the higher learning jumps of classification curve.</p>
<table-wrap id="table-2"><label>Table 2</label><caption><title>Dynamic learning rate</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Epochs</th>
<th align="left">Learning rate</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">1&#x2013;500</td>
<td align="left">0.001</td>
</tr>
<tr>
<td align="left">501&#x2013;800</td>
<td align="left">0.0001</td>
</tr>
<tr>
<td align="left">801&#x2013;1000</td>
<td align="left">0.00001</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To handle the over fitting dropout is also used to randomize the feature extraction and selection. We try different values of dropout to check where the best features are selected and extracted. The train and test loss along the dropout ratio is shown in <xref ref-type="table" rid="table-3">Tab. 3</xref>.</p>
<table-wrap id="table-3"><label>Table 3</label><caption><title>Dynamic train loss and test loss</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Dropout ratio</th>
<th align="left">Train loss</th>
<th align="left">Test loss</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">0.4</td>
<td align="left">0.00156</td>
<td align="left">0.0325</td>
</tr>
<tr>
<td align="left">0.5</td>
<td align="left">0.00164</td>
<td align="left">0.00173</td>
</tr>
<tr>
<td align="left">0.6</td>
<td align="left">0.00982</td>
<td align="left">0.0145</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We observe for dropout ratio 0.5 model gives us best loss values. The values of train and test loss are much near than for other dropout ratio. So, we continue our whole training for dropout ratio 0.5. The training process for 1000 epochs with this hardware take 50 h to complete [<xref ref-type="bibr" rid="ref-25">25</xref>].</p>
</sec>
<sec id="s4_2"><label>4.2</label><title>Challenges</title>
<p>SGD optimizer navigating to global minimum loss by taking step towards where the loss decreases.</p>
<p>sBut sometimes the SGD stuck into local minima because there is no next point there the loss recrudesces. So. It stuck in local minima as shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref> and optimization stops.</p>
<p>This problem is solved by momentum. The momentum accelerates the SGD to move in the desired direction as shown in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>.</p>
<fig id="fig-6"><label>Figure 6</label><caption><title>Optimization without momentum</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-6.png"/></fig>
<p>Does momentum work by adding the fraction (lambda) if the last update vector in the currently updated vector as explained in <xref ref-type="disp-formula" rid="eqn-5">Eq. (5)</xref> and <xref ref-type="disp-formula" rid="eqn-6">Eq. (6)</xref>.
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03B7;</mml:mi><mml:mrow><mml:mtext>&#xA0;</mml:mtext></mml:mrow><mml:mrow><mml:msub><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mi>&#x03B8;</mml:mi></mml:msub></mml:mrow><mml:mspace width="thickmathspace" /><mml:mi>J</mml:mi><mml:mspace width="thickmathspace" /><mml:mrow><mml:mo>(</mml:mo><mml:mi>&#x03B8;</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mi>&#x03B8;</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x03B8;</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mi>t</mml:mi></mml:msub></mml:mrow></mml:math></disp-formula></p>
<p>While experiments many people use different values. The most common value of the (lambda) for momentum is 0.9 or near to this. After 1000 epochs the values of MSE shown in graph in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>. In <xref ref-type="fig" rid="fig-7">Fig. 7</xref> we clearly show that this technique of scaling significantly enhances the efficiency as compared to single dimension scaling method.</p>
<fig id="fig-7"><label>Figure 7</label><caption><title>Optimization with momentum</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-7.png"/></fig>
<fig id="fig-8"><label>Figure 8</label><caption><title>Mean square error graph</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-8.png"/></fig>
<p>We have used real time data in this research like Mock up exercise (demonstration data) of images and videos on several types [<xref ref-type="bibr" rid="ref-20">20</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>].</p>
</sec>
</sec>
<sec id="s5"><label>5</label><title>Results</title>
<sec id="s5_1"><label>5.1</label><title>Confusion Matrix</title>
<p>In this confusion matrix shown in <xref ref-type="fig" rid="fig-9">Fig. 9</xref>, we use 15000 plus images and videos of live data and the result of our research is better than previous research. Through efficient-net the result better results than previous one and use local and global data and on live videos. In Confusion matrix show the accuracy and loss rate of training and validation.</p>
<fig id="fig-9"><label>Figure 9</label><caption><title>Confusion matrix</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-9.png"/></fig>
<p><xref ref-type="table" rid="table-4">Tab. 4</xref> shows result comparison of the proposed model with state-of-the-art deep learning algorithms. Transfer learning on same dataset through faster RCNN with ZFNet and VGG-16 accurately identified weapons with only &#x0025; and &#x0025; of accuracy. On the other hand Yolov3 and YOLO v4 achieved &#x0025; and &#x0025; of accuracy. The comparison shows that the proposed model outperformed all the previous approaches with 98.12&#x0025; accuracy.</p>
<table-wrap id="table-4"><label>Table 4</label><caption><title>Model comparison</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">Model</th>
<th align="left">F1 score</th>
<th align="left">Precision</th>
<th align="left">Recall</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">Faster RCNN</td>
<td align="left">0.9301</td>
<td align="left">0.9611</td>
<td align="left">0.9414</td>
</tr>
<tr>
<td align="left">Yolov4<break/>Proposed model</td>
<td align="left">0.9321<break/>0.9883</td>
<td align="left">0.9134<break/>1.0</td>
<td align="left">0.9266<break/>0.9712</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5_2"><label>5.2</label><title>Accuracy</title>
<p>As we increase the ratio of session data the accuracy increases and loss ratio decreases as comparative to previous observation, shown in <xref ref-type="fig" rid="fig-10">Fig. 10</xref>. The accuracy rate increase increases when the Epochs rate increases.</p>
<p>In <xref ref-type="fig" rid="fig-10">Fig. 10</xref> the Epoch size increase and total Epoch in this fig is1500. When Epochs size increase and the accuracy also increase.</p>
</sec>
<sec id="s5_3"><label>5.3</label><title>Loss</title>
<p>The <xref ref-type="fig" rid="fig-11">Fig. 11</xref> illustrates the training and validation loss decreasing trend similar to mean squared error.</p>
<p>The loss rate decreases when the value of Epochs increases shown in <xref ref-type="fig" rid="fig-11">Fig. 11</xref>. BY using Efficient-Net the loss rate low as compared to other Algorithms This figure show the training and validation loss and also show Epochs size decrease and the loss rate of training and validation.</p>
<fig id="fig-10"><label>Figure 10</label><caption><title>Accuracy ratio</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-10.png"/></fig>
<fig id="fig-11"><label>Figure 11</label><caption><title>Training and validation loss</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_27571-fig-11.png"/></fig>
</sec>
</sec>
<sec id="s6"><label>6</label><title>Conclusion</title>
<p>In this paper, we presented Efficient-net used for object detection. Most of the object detection is done with CNN based neural networks in the modern era. The basic motivation behind this Efficient-net base weapon detection system that employee&#x0027;s region proposal network was to reduce the false positive and make a model with near to real time efficiency that is mainly the center of attention of the researchers.</p>
<p>For increasing the robustness and reducing the false positive of the model we gathered local data set and live videos of cameras that&#x0027;s we defined. We trained these models and used pre-trained feature extractors because it saves a lot of time to fine tune a model according to our problem. Experiments show that our model obtained better results for weapon detection system than the previous research. By using Efficient-Net algorithms the better accuracy achieved as compared to other algorithms.</p>
<p>We will extend our model to cover move objects more efficiently and will drive it to give our own model.</p>
</sec>
</body>
<back>
<ack>
<p>We thank our families and colleagues who provided us with moral support.</p>
</ack>
<fn-group>
<fn fn-type="other"><p><bold>Funding Statement:</bold> The authors received no specific funding for this study.</p></fn>
<fn fn-type="conflict"><p><bold>Conflicts of Interest:</bold> The authors declare that they have no conflicts of interest to report regarding the present study.</p></fn>
</fn-group>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Alexandrie</surname></string-name></person-group>, &#x201C;<article-title>Surveillance cameras and crime: A review of randomized and natural experiments</article-title>,&#x201D; <source>Journal of Scandinavian Studies in Criminology and Crime Prevention</source>, vol. <volume>18</volume>, no. <issue>2</issue>, pp. <fpage>210</fpage>&#x2013;<lpage>222</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. P.</given-names> <surname>Ashby</surname></string-name></person-group>, &#x201C;<article-title>The value of CCTV surveillance cameras as an investigative tool: An empirical analysis</article-title>,&#x201D; <source>European Journal on Criminal Policy and Research</source>, vol. <volume>23</volume>, no. <issue>3</issue>, pp. <fpage>441</fpage>&#x2013;<lpage>459</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>LeCun</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Bengio</surname></string-name> and <string-name><given-names>G.</given-names> <surname>Hinton</surname></string-name></person-group>, &#x201C;<article-title>Deep learning</article-title>,&#x201D; in <conf-name>Int. Society for Behavioral Neuroscience (ISBN) Conf.</conf-name>, <conf-loc>Wuhan, China</conf-loc>, pp. <fpage>436</fpage>&#x2013;<lpage>444</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>He</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ren</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Sun</surname></string-name></person-group>, &#x201C;<article-title>Delivering deep into rectifiers: surpassing human-level performance on imagenet classification</article-title>,&#x201D; in <conf-name>IEEE Int. Conf. on Computer Vision (ICCV)</conf-name>, <conf-loc>Santiago, Chile</conf-loc>, pp. <fpage>1026</fpage>&#x2013;<lpage>1034</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Bagavathi</surname></string-name> and <string-name><given-names>O.</given-names> <surname>Saraniya</surname></string-name></person-group>, &#x201C;<article-title>Hardware designs for histogram of oriented gradients in pedestrian detection: A survey</article-title>,&#x201D; in <conf-name>Int. Conf. on Advanced Computing and Communicating Systems (ICACCS)</conf-name>, <conf-loc>Coimbatore, India</conf-loc>, pp. <fpage>849</fpage>&#x2013;<lpage>854</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Anguelov</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Erhan</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Szegedy</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Reed</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>SSD: Single shot multibox detector</article-title>,&#x201D; in <conf-name>European Conf. on Computer Vision (ECCV)</conf-name>, <conf-loc>Amsterdam, Netherlands</conf-loc>, pp. <fpage>21</fpage>&#x2013;<lpage>37</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>He</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ren</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Sun</surname></string-name></person-group>, &#x201C;<article-title>Spatial pyramid pooling in deep convolutional networks for visual recognition</article-title>,&#x201D; <source>IEEE Transactions on Pattern Analysis and Machine Intelligence</source>, vol. <volume>37</volume>, no. <issue>9</issue>, pp. <fpage>1904</fpage>&#x2013;<lpage>1916</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Abruzzo</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Carey</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Lowrance</surname></string-name>, <string-name><given-names>E.</given-names> <surname>Sturzinger</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Arnold</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Cascaded neural networks for identification and posture-based threat assessment of armed people</article-title>,&#x201D; in <conf-name>IEEE, Int. Symp. on Technologies for Homeland Security (HST)</conf-name>, <conf-loc>Woburn, MA, USA</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>7</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Dhillon</surname></string-name> and <string-name><given-names>G. K.</given-names> <surname>Verma</surname></string-name></person-group>, &#x201C;<article-title>Convolutional neural network: A review of models, methodologies and applications to object detection</article-title>,&#x201D; <source>Progress in Artificial Intelligence</source>, vol. <volume>9</volume>, no. <issue>2</issue>, pp. <fpage>85</fpage>&#x2013;<lpage>112</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>G. K.</given-names> <surname>Verma</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Dhillon</surname></string-name></person-group>, &#x201C;<article-title>A handheld gun detection using faster R-CNN deep learning</article-title>,&#x201D; in <conf-name>Int. Conf. on Computer and Communication Technology (ICCCT)</conf-name>, <conf-loc>Allahabad, India</conf-loc>, pp. <fpage>84</fpage>&#x2013;<lpage>88</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Lim</surname></string-name>, <string-name><given-names>M. I.</given-names> <surname>Al Jobayer</surname></string-name>, <string-name><given-names>V. M.</given-names> <surname>Baskaran</surname></string-name>, <string-name><given-names>J. M.</given-names> <surname>Lim</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Wong</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Gun detection in surveillance videos using deep neural networks</article-title>,&#x201D; in <conf-name>Asia-Pacific Signal and Information Processing Association Annual Summit and Conf. (APSIPA ASC)</conf-name>, <conf-loc>Lanzhou, China</conf-loc>, pp. <fpage>1998</fpage>&#x2013;<lpage>2002</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D. M.</given-names> <surname>El-Din</surname></string-name> and <string-name><given-names>A. E.</given-names> <surname>Hassanien</surname></string-name></person-group>, &#x201C;<article-title>Multi-spectral fusion system based on deep transfer learning and dempster-shafer theory</article-title>,&#x201D; <source>Journal of Theoretical and Applied Information Technology</source>, vol. <volume>98</volume>, no. <issue>6</issue>, pp. <fpage>1817</fpage>&#x2013;<lpage>3195</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Jain</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Vikram</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Kashyap</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Jain</surname></string-name></person-group>, &#x201C;<article-title>Weapon detection using artificial intelligence and deep learning for security</article-title>,&#x201D; in <conf-name>Int. Conf. on Electronics and Sustainable Communication Systems (ICESC)</conf-name>, <conf-loc>Coimbatore, India</conf-loc>, pp. <fpage>193</fpage>&#x2013;<lpage>198</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>U. V.</given-names> <surname>Navalgund</surname></string-name> and <string-name><given-names>K.</given-names> <surname>Priyadharshini</surname></string-name></person-group>, &#x201C;<article-title>Crime intention detection system using deep learning</article-title>,&#x201D; in <conf-name>Int. Conf. on Circuits and Systems in Digital Enterprise Technology (ICCSDET)</conf-name>, <conf-loc>Kottayam, India</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>6</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Raturi</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Rani</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Madan</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Dosanjh</surname></string-name></person-group>, &#x201C;<article-title>ADoCW: An automated method for detection of concealed weapon</article-title>,&#x201D; in <conf-name>Fifth Int. Conf. on Image Information Processing (ICIIP)</conf-name>, <conf-loc>Shimla, India</conf-loc>, pp. <fpage>181</fpage>&#x2013;<lpage>186</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Mahajan</surname></string-name> and <string-name><given-names>D.</given-names> <surname>Padha</surname></string-name></person-group>, &#x201C;<article-title>Detection of concealed weapons using image processing techniques: A review</article-title>,&#x201D; in <conf-name>First Int. Conf. on Secure Cyber Computing and Communication (ICSCCC)</conf-name>, <conf-loc>Jalandhar, India</conf-loc>, pp. <fpage>375</fpage>&#x2013;<lpage>378</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Gelana</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Yadav</surname></string-name></person-group>, &#x201C;<chapter-title>Firearm detection from surveillance cameras using image processing and machine learning techniques</chapter-title>,&#x201D; in <source>Smart Innovations in Communication and Computational Sciences</source>, <publisher-name>Springer</publisher-name>, <publisher-loc>Singapore</publisher-loc>, pp. <fpage>25</fpage>&#x2013;<lpage>34</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Z. Q.</given-names> <surname>Zhao</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Zheng</surname></string-name>, <string-name><given-names>S. T.</given-names> <surname>Xu</surname></string-name> and <string-name><given-names>X.</given-names> <surname>Wu</surname></string-name></person-group>, &#x201C;<article-title>Object detection with deep learning: A review</article-title>,<italic>&#x201D;</italic> <source>IEEE Transactions on Neural Networks and Learning Systems</source>, vol. <volume>30</volume>, no. <issue>11</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>21</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>S. H.</given-names> <surname>Tsang</surname></string-name></person-group>, &#x201C;<article-title>Review-NAS-FPN: Learning scalable feature pyramid architecture for object detection</article-title>,&#x201D; in <conf-name>IEEE Conf. on Computer Vision and Pattern Recognition</conf-name>, <conf-loc>Long beach, CA, USA</conf-loc>, pp. <fpage>7036</fpage>&#x2013;<lpage>7045</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>E.</given-names> <surname>Arif</surname></string-name>, <string-name><given-names>S. K.</given-names> <surname>Shahzad</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Mustafa</surname></string-name>, <string-name><given-names>M. A.</given-names> <surname>Jaffar</surname></string-name> and <string-name><given-names>M. W.</given-names> <surname>Iqbal</surname></string-name></person-group>, &#x201C;<article-title>Deep neural networks for gun detection in public surveillance</article-title>,&#x201D; <source>Intelligent Automation and Soft Computing (IASC)</source>, vol. <volume>32</volume>, no. <issue>2</issue>, pp. <fpage>909</fpage>&#x2013;<lpage>922</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Cheng</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Liu</surname></string-name> and <string-name><given-names>W.</given-names> <surname>Xing</surname></string-name></person-group>, &#x201C;<article-title>Weighted feature fusion and attention mechanism for object detection</article-title>,&#x201C; <source>Journal of Electronic Imaging</source>, vol. <volume>30</volume>, no. <issue>2</issue>, pp. <fpage>23015</fpage>&#x2013;<lpage>23031</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M. S.</given-names> <surname>Hosseini</surname></string-name>, <string-name><given-names>J. S.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Fu</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Su</surname></string-name></person-group>, &#x201C;<article-title>CONET: Channel optimization for convolutional neural networks</article-title>,&#x201D; in <conf-name>IEEE Conf. on Computer Vision and Pattern Recognition</conf-name>, <conf-loc>Nashville, TN, USA</conf-loc>, pp. <fpage>326</fpage>&#x2013;<lpage>335</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Nazir</surname></string-name>, <string-name><given-names>M. W.</given-names> <surname>Iqbal</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Alyas</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Hamid</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Saleem</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Load balancing framework for cross-region tasks in cloud computing</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>70</volume>, no. <issue>1</issue>, pp. <fpage>1479</fpage>&#x2013;<lpage>1490</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C. C.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>J. Y.</given-names> <surname>Ba</surname></string-name>, <string-name><given-names>T. J.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>C. C.</given-names> <surname>Chan</surname></string-name>, <string-name><given-names>K. C.</given-names> <surname>Wang</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Efficient net: A low-bandwidth IoT image sensor framework for cassava leaf disease classification</article-title>,&#x201D; <source>Sensors and Materials</source>, vol. <volume>33</volume>, no. <issue>11</issue>, pp. <fpage>4031</fpage>&#x2013;<lpage>4044</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C. Y.</given-names> <surname>Luo</surname></string-name>, <string-name><given-names>S. Y.</given-names> <surname>Cheng</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Xu</surname></string-name> and <string-name><given-names>P.</given-names> <surname>Li</surname></string-name></person-group>, &#x201C;<article-title>Human behavior recognition model based on improved effivient net</article-title>,&#x201D; <source>Procedia Computer Science</source>, vol. <volume>199</volume>, no. <issue>1</issue>, pp. <fpage>369</fpage>&#x2013;<lpage>376</lpage>, <year>2022</year>.</mixed-citation></ref>
</ref-list>
</back>
</article>