<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">34720</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2023.034720</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Multi-Classification of Polyps in Colonoscopy Images Based on an Improved Deep Convolutional Neural Network</article-title>
<alt-title alt-title-type="left-running-head">Multi-Classification of Polyps in Colonoscopy Images Based on an Improved Deep Convolutional Neural Network</alt-title>
<alt-title alt-title-type="right-running-head">Multi-Classification of Polyps in Colonoscopy Images Based on an Improved Deep Convolutional Neural Network</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Liu</surname><given-names>Shuang</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref>
<xref ref-type="aff" rid="aff-2">2</xref>
<xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Liu</surname><given-names>Xiao</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Chang</surname><given-names>Shilong</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Sun</surname><given-names>Yufeng</given-names>
</name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Li</surname><given-names>Kaiyuan</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Hou</surname><given-names>Ya</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Wang</surname><given-names>Shiwei</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>Meng</surname><given-names>Jie</given-names>
</name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-9" contrib-type="author">
<name name-style="western"><surname>Zhao</surname><given-names>Qingliang</given-names>
</name><xref ref-type="aff" rid="aff-6">6</xref></contrib>
<contrib id="author-10" contrib-type="author">
<name name-style="western"><surname>Wu</surname><given-names>Sibei</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-11" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Yang</surname><given-names>Kun</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref>
<xref ref-type="aff" rid="aff-2">2</xref>
<xref ref-type="aff" rid="aff-3">3</xref><email>yangkun@hbu.edu.cn</email></contrib>
<contrib id="author-12" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Xue</surname><given-names>Linyan</given-names>
</name><xref ref-type="aff" rid="aff-1">1</xref>
<xref ref-type="aff" rid="aff-2">2</xref>
<xref ref-type="aff" rid="aff-3">3</xref><email>lyxue@hbu.edu.cn</email></contrib>
<aff id="aff-1"><label>1</label><institution>College of Quality and Technical Supervision, Hebei University</institution>, <addr-line>Baoding, 071002</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>Hebei Technology Innovation Center for Lightweight of New Energy Vehicle Power System</institution>, <addr-line>Baoding, 071002</addr-line>, <country>China</country></aff>
<aff id="aff-3"><label>3</label><institution>National &#x0026; Local Joint Engineering Research Center of Metrology Instrument and System, Hebei University</institution>, <addr-line>Baoding, 071002</addr-line>, <country>China</country></aff>
<aff id="aff-4"><label>4</label><institution>College of Electronic Information Engineering, Hebei University</institution>, <addr-line>Baoding, 071002</addr-line>, <country>China</country></aff>
<aff id="aff-5"><label>5</label><institution>Department of Orthopedics, Affiliated Hospital of Hebei University</institution>, <addr-line>Baoding, 071002</addr-line>, <country>China</country></aff>
<aff id="aff-6"><label>6</label><institution>State Key Laboratory of Molecular Vaccinology and Molecular Diagnostics, Center for Molecular Imaging and Translational Medicine, Department of Laboratory Medicine, School of Public Health, Xiamen University</institution>, <addr-line>Xiamen, 361102</addr-line>, <country>China</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Authors: Kun Yang. Email: <email>yangkun@hbu.edu.cn</email>; Linyan Xue. Email: <email>lyxue@hbu.edu.cn</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic"><year>2023</year></pub-date>
<pub-date date-type="pub" publication-format="electronic"><day>1</day><month>5</month><year>2023</year>
</pub-date>
<volume>75</volume>
<issue>3</issue>
<fpage>5837</fpage>
<lpage>5852</lpage>
<history>
<date date-type="received"><day>25</day><month>7</month><year>2022</year>
</date>
<date date-type="accepted"><day>20</day><month>10</month><year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2023 Liu et al.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Liu et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_34720.pdf"></self-uri>
<abstract>
<p>Achieving accurate classification of colorectal polyps during colonoscopy can avoid unnecessary endoscopic biopsy or resection. This study aimed to develop a deep learning model that can automatically classify colorectal polyps histologically on white-light and narrow-band imaging (NBI) colonoscopy images based on World Health Organization (WHO) and Workgroup serrAted polypS and Polyposis (WASP) classification criteria for colorectal polyps. White-light and NBI colonoscopy images of colorectal polyps exhibiting pathological results were firstly collected and classified into four categories: conventional adenoma, hyperplastic polyp, sessile serrated adenoma/polyp (SSAP) and normal, among which conventional adenoma could be further divided into three sub-categories of tubular adenoma, villous adenoma and villioustublar adenoma, subsequently the images were re-classified into six categories. In this paper, we proposed a novel convolutional neural network termed Polyp-DedNet for the four- and six-category classification tasks of colorectal polyps. Based on the existing classification network ResNet50, Polyp-DedNet adopted dilated convolution to retain more high-dimensional spatial information and an Efficient Channel Attention (ECA) module to improve the classification performance further. To eliminate gridding artifacts caused by dilated convolutions, traditional convolutional layers were used instead of the max pooling layer, and two convolutional layers with progressively decreasing dilation were added at the end of the network. Due to the inevitable imbalance of medical image data, a regularization method DropBlock and a Class-Balanced (CB) Loss were performed to prevent network overfitting. Furthermore, the 5-fold cross-validation was adopted to estimate the performance of Polyp-DedNet for the multi-classification task of colorectal polyps. Mean accuracies of the proposed Polyp-DedNet for the four- and six-category classifications of colorectal polyps were 89.91% &#x00B1; 0.92% and 85.13% &#x00B1; 1.10%, respectively. The metrics of precision, recall and F1-score were also improved by 1%&#x007E;2% compared to the baseline ResNet50. The proposed Polyp-DedNet presented state-of-the-art performance for colorectal polyp classifying on white-light and NBI colonoscopy images, highlighting its considerable potential as an AI-assistant system for accurate colorectal polyp diagnosis in colonoscopy.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Colorectal polyps</kwd>
<kwd>four- and six-category classifications</kwd>
<kwd>convolutional neural network</kwd>
<kwd>dilated residual network</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>Research Fund for Foundation of Hebei University</funding-source>
<award-id>DXK201914</award-id>
</award-group>
<award-group id="awg2">
<funding-source>President of Hebei University</funding-source>
<award-id>XZJJ201914</award-id>
</award-group>
<award-group id="awg3">
<funding-source>Post-graduate&#x2019;s Innovation Fund Project of Hebei University</funding-source>
<award-id>HBU2022SS003</award-id>
</award-group>
<award-group id="awg4">
<funding-source>Special Project for Cultivating College Students&#x0027; Scientific and Technological Innovation Ability in Hebei Province</funding-source>
<award-id>22E50041D</award-id>
</award-group>
<award-group id="awg5">
<funding-source>Guangdong Basic and Applied Basic Research Foundation</funding-source>
<award-id>2021A1515011654</award-id>
</award-group>
<award-group id="awg6">
<funding-source>Central Universities of China</funding-source>
<award-id>20720210117</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Based on 2020 reports, Colorectal cancer (CRC) is a significant public health problem, the second leading cause of cancer-related death worldwide and the fifth leading cause of cancer-related death in China [<xref ref-type="bibr" rid="ref-1">1</xref>]. Colonoscopy is one of the most important and effective methods for the early detection and resection of colorectal neoplasms, which has been adopted in many countries to improve the detection rate of adenoma and reduce CRC mortality [<xref ref-type="bibr" rid="ref-2">2</xref>&#x2013;<xref ref-type="bibr" rid="ref-4">4</xref>]. Studies have shown that about 85% of CRC is derived from adenomas, and endoscopic resection of colorectal polyps (CP) can reduce the incidence of CRC [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-6">6</xref>]. However, the results of colonoscopy tend to be affected by the doctor&#x0027;s clinical experience, fatigue and other subjective factors, which in turn, the diagnostic performance among clinicians is inconsistent [<xref ref-type="bibr" rid="ref-7">7</xref>]. Moreover, a few neoplastic lesions remain challenging to diagnose accurately, even for expert endoscopists [<xref ref-type="bibr" rid="ref-8">8</xref>].</p>
<p>To resolve these problems, researchers have been working to employ computerized methods. Computer-aided diagnosis (CAD) nowadays plays a significant role in clinical research and practice which began to develop in the mid-1980s and was first used for chest radiography and mammography for cancer detection and diagnosis [<xref ref-type="bibr" rid="ref-9">9</xref>]. Machine learning (ML)-based CAD techniques are characterized by hand-crafted features which rely heavily on expert experience. Therefore, these classification and detection methods illustrate poor performance in generalization. With the rapid development of artificial intelligence technology, deep neural networks with end-to-end learning have been increasingly exploited to design CAD systems for the automated diagnosis of medical diseases [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-11">11</xref>], including the auxiliary diagnosis of colorectal polyps.</p>
<p>Colorectal polyps can be categorized into conventional adenoma, hyperplastic polyp, sessile serrated adenoma/polyp (SSAP) based on the Workgroup serrAted polypS and Polyposis (WASP) classification criteria [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-13">13</xref>]. However, according to the World Health Organization (WHO) classification criteria for colorectal polyps, conventional adenomas are divided into tubular adenoma, villous adenoma, and villioustublar adenoma. Each type of polyps has a different chance of developing into CRC [<xref ref-type="bibr" rid="ref-12">12</xref>]. For example, several studies have shown that conventional adenomas and SSAP have different pathways to cancer but have a similar relatively high risk of developing CRC. On the contrary, the hyperplastic polyp can hardly develop into CRC [<xref ref-type="bibr" rid="ref-12">12</xref>,<xref ref-type="bibr" rid="ref-14">14</xref>].</p>
<p>We propounded a new deep learning network, Polyp-DedNet, to achieve more accurate four- and six-category classifications of colorectal polyps according to the WHO and WASP criteria during white light and narrow-band imaging colonoscopy. The four-category task divided the images into conventional adenoma, SSAP, hyperplastic polyp and normal. According to WHO criteria, our proposed network can further predict a colonoscopy image into one of the six categories: SSAP, hyperplastic polyp, tubular adenoma, villous adenoma, villioustublar adenoma and normal. Polyp-DedNet adopts the dilated residual network to retain more spatial information for improving image classification accuracy, an attentional mechanism to focus on the lesion area in the image and the regularization method named DropBlock [<xref ref-type="bibr" rid="ref-15">15</xref>] to prevent network overfitting. In the training process, Polyp-DedNet was performed as the basic network of colorectal polyp classification, and the transfer learning method was presented to improve the accuracy and rapidity of network training. Meanwhile, the Class-Balanced (CB) Loss [<xref ref-type="bibr" rid="ref-16">16</xref>] was used to address problems in imbalanced data learning.</p>
<p>The major contributions of our work are as follows: (1) A novel framework was proposed for WASP and WHO pathological classification under white light and narrow-band light, which increased the diversity of classification categories compared to previous studies on colon polyp classification; (2) We introduced the dilated convolution and the improved attention mechanism into the residual block to obtain more feature information of small polyps. And the effect of data imbalance in our collected datasets was alleviated by CB Loss; (3) The proposed Polyp-DedNet was efficient on four- and six-category classification of colorectal polyps and significantly superior to other state-of-the-art classification networks. A more diverse and accurate multi-classification network based on the WASP and WHO classification criteria can assist doctors in colonoscopy, avoid unnecessary resection and detect adenomatous polyps as early as possible.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Work</title>
<p>Machine learning (ML)-based CAD techniques have been widely used in colorectal classification tasks. For example, Shin et al. [<xref ref-type="bibr" rid="ref-17">17</xref>] proposed a dictionary-based learning scheme, using support vector machines (SVM) for the patch-level images and a simple threshold method for the whole image to classify polyp and normal (no polyps) images. Tamaki et al. [<xref ref-type="bibr" rid="ref-18">18</xref>] presented a scale-invariant feature transform (SIFT) algorithm to extract local features of colonoscopy images and implemented SVM with radial basis function (RBF) kernel to classify hyperplastic polyp, tubular adenoma, and cancer under narrow-band light, achieving an accuracy rate of 94.1%. However, the feature extraction process of ML-based CAD is difficult due to the influence of limited illumination conditions, blurred fields and variations in viewpoint [<xref ref-type="bibr" rid="ref-19">19</xref>].</p>
<p>Nowadays, unlike the manual feature extraction of traditional ML networks, deep convolutional neural networks (DCNN) can automatically extract features to complete the task of classifying colorectal polyps. Wang et al. [<xref ref-type="bibr" rid="ref-20">20</xref>] combined the global average pooling (GAP) and the classical deep learning network ResNet to design a ResNet-GAP network to classify polyps and normal under white light, with a test accuracy of 98%. Chen et al. [<xref ref-type="bibr" rid="ref-21">21</xref>] established a computer-aided system named DNN-CAD based on NBI to classify adenomatous and hyperplastic polyps smaller than 5 mm with a sensitivity of 96.3% and a shorter classification time than experts. Byrne et al. [<xref ref-type="bibr" rid="ref-22">22</xref>] designed a deep learning model based on the Inception network to achieve the real-time classification of diminutive adenomas and hyperplastic polyps based on NBI, and the model&#x2019;s accuracy was 94%. Ozawa et al. [<xref ref-type="bibr" rid="ref-23">23</xref>] adopted Single Shot Multi Box Detector (SSD) network to build an automatic multi-classification and detection system for colorectal polyps under white light, achieving a classification accuracy of 83%. In their study, the polyp types include adenoma, hyperplastic polyp, SSAP, cancer, the others and normal. Wang et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] used ResNet50 to classify the four conditions of polyps, inflammation, tumor and normal under white light, obtaining an accuracy of 94.48%. Most studies have focused on improving deep learning network structures for binary classification and detection of adenomas/non-adenomas, or polyps/normal. However, very limited studies were done on the multi-classification of colorectal polyps according to both WASP and WHO criteria [<xref ref-type="bibr" rid="ref-12">12</xref>]. Furthermore, most previous studies ignored the data imbalance in the colon polyp multi-classification datasets, which may lead to a falsely high overall classification accuracy.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Methods</title>
<p>ResNet is an effective convolutional neural network that solves the degradation problem and reduces the difficulty of deep network training [<xref ref-type="bibr" rid="ref-25">25</xref>]. As shown in <xref ref-type="fig" rid="fig-1">Fig. 1a</xref>, ResNet50 mainly consists of two residual modules: Conv Block and ID Block. To obtain a large receptive field, images and feature maps are down-sampled by the pooling layer (Max pool) of stage 1 and each convolutional layer with large strides in Conv Blocks of stages 2 to 5. However, the resolution of the image or feature map is reduced after down-sampling, leaving only less spatial information, which will result in the information loss of some small adenomas in colorectal images and performance decrease in polyp classification and localization. In this paper, we build an improved Polyp-DedNet based on ResNet50, which is shown in <xref ref-type="fig" rid="fig-1">Fig. 1b</xref>. In Polyp-DedNet, dilated convolutions with the dilated factors of 2 and 4 are respectively applied in IDE block and Conv Block of stages 4 and 5 to keep the spatial resolution for deep neural network without costing too much time and memory, reduce the down-sampling factor and obtain an effective large receptive field. Efficient Channel Attention (ECA) modules are also used in IDE blocks and Conv Blocks of stages 2 to 5 to further improve the classification performance of the network. Furthermore, to eliminate gridding artifacts caused by dilated convolutions, two traditional convolutional layers are applied in stage 1 rather than the max pooling layer, and two dilated convolutional layers (ConvE-layer) with progressively decreasing dilation are added at the end of stage 5. Due to the inevitable imbalance of medical image data, a regularization method named DropBlock and the CB loss are performed to prevent network overfitting.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>The architecture of (a) ResNet50 and (b) Polyp-DedNet. Here, d represents the dilation factor with the value of 4, 2 or 1</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_34720-fig-1.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Dilated Residual Networks</title>
<p>The detail description of ResNet50 is formed from five stages of convolutional layers presented in the left three columns of <xref ref-type="table" rid="table-1">Table 1</xref>. The first layer in each stage performs a down-sampling process by striding. We define <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> to denote the <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:msup><mml:mi>i</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mi>h</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> layer in stage <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mi>j</mml:mi></mml:math></inline-formula>, where <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mi>j</mml:mi></mml:math></inline-formula> &#x003D;1,&#x2026;,5. The output of <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> of the stage in ResNet50 network is formulated as</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Detail description of ResNet50 and the proposed dilated residual network used in our study</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Layer</th>
<th>ResNet50</th>
<th>Output size</th>
<th>Proposed dilated residual network</th>
<th>Output size</th>
</tr>
</thead>
<tbody>
<tr>
<td>Conv1</td>
<td>7 &#x00D7; 7, 64, stride 2</td>
<td>112 &#x00D7; 112</td>
<td>7 &#x00D7; 7, 16, stride 1</td>
<td>112 &#x00D7; 112</td>
</tr>
<tr>
<td></td>
<td>3 &#x00D7; 3 max pooling stride 2</td>
<td></td>
<td>3 &#x00D7; 3, 16, stride 1<break/>3 &#x00D7; 3, 32, stride 2</td>
<td></td>
</tr>
<tr>
<td>Conv2_x</td>
<td><inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>64</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>64</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>256</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula></td>
<td>56 &#x00D7; 56</td>
<td><inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>64</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>64</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>256</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula></td>
<td>56 &#x00D7; 56</td>
</tr>
<tr>
<td>Conv3_x</td>
<td><inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>128</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>128</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>512</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mn>4</mml:mn></mml:math></inline-formula></td>
<td>28 &#x00D7; 28</td>
<td><inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>128</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>128</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>512</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mn>4</mml:mn></mml:math></inline-formula></td>
<td>28 &#x00D7; 28</td>
</tr>
<tr>
<td>Conv4_x</td>
<td><inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>256</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>256</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1024</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mn>6</mml:mn></mml:math></inline-formula></td>
<td>14 &#x00D7; 14</td>
<td><inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>256</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>256</mml:mn><mml:mspace width="thinmathspace" /><mml:mspace width="thinmathspace" /><mml:mrow><mml:mtext>dilation</mml:mtext></mml:mrow><mml:mspace width="thinmathspace" /><mml:mspace width="thinmathspace" /><mml:mn>2</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1024</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mn>6</mml:mn></mml:math></inline-formula></td>
<td>28 &#x00D7; 28</td>
</tr>
<tr>
<td>Conv5_x</td>
<td><inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>512</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>512</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2048</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula></td>
<td>7 &#x00D7; 7</td>
<td><inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:mrow><mml:mo>(</mml:mo><mml:mtable columnalign="left" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>512</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>512</mml:mn><mml:mspace width="thinmathspace" /><mml:mspace width="thinmathspace" /><mml:mrow><mml:mtext>dilation</mml:mtext></mml:mrow><mml:mspace width="thinmathspace" /><mml:mspace width="thinmathspace" /><mml:mn>4</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2048</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula><break/>3 &#x00D7; 3, 512, dilation 2<break/>3 &#x00D7; 3, 512, dilation 1</td>
<td>28 &#x00D7; 28</td>
</tr>
<tr>
<td>FC</td>
<td colspan="4">AvgPool, number of classes fc, softmax</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x2217;</mml:mo><mml:msubsup><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo>&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>+</mml:mo><mml:mi>b</mml:mi><mml:mo>=</mml:mo><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mi>a</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:msubsup><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mi>b</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>where <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:msubsup><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> represents the filter combined with layer <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and the domain of <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mi>p</mml:mi></mml:math></inline-formula> denotes the feature map in <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>.</p>
<p>To increase the resolution of output feature maps without reducing the receptive field of each output unit, dilated convolution is adopted in the final two stages, <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, in ResNet50 to construct the dilated residual network, which is shown in the right two columns of <xref ref-type="table" rid="table-1">Table 1</xref>.</p>

<p>For <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, we remove the striding in <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> and substitute the convolution operators with dilated convolutions with a dilated factor of 2, which is represented as:</p>
<p><disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msubsup><mml:msub><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msubsup><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo>&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>b</mml:mi><mml:mo>=</mml:mo><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mi>a</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:msubsup><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mi>b</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>where all <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mi>i</mml:mi><mml:mo>&#x2265;</mml:mo><mml:mn>1.</mml:mn></mml:math></inline-formula> For the layer of <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:msup><mml:mi>S</mml:mi><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>, the striding in <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> is also removed. And the dilated convolutions with a dilated factor of 4 are applied to compensating for the reduction in the receptive field due to removing subsampling, which can be formulated as</p>
<p><disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msubsup><mml:msub><mml:mo>&#x2217;</mml:mo><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub><mml:msubsup><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo>&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>+</mml:mo><mml:mn>4</mml:mn><mml:mi>b</mml:mi><mml:mo>=</mml:mo><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mi>a</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:msubsup><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>5</mml:mn></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mi>b</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>where all <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mi>i</mml:mi><mml:mo>&#x2265;</mml:mo><mml:mn>1.</mml:mn></mml:math></inline-formula></p>
<p>However, employing dilated convolution may cause gridding artifacts, especially when a feature map has higher-frequency content than the sampling rate of the dilated convolution. Additionally, the max pooling layer after the initial 7 &#x00D7; 7 convolution, which leads to high-amplitude high-frequency activations, will also eventually aggravate gridding artifacts. In this case, we further modify the dilated residual network and replace the maximum pooling with convolution layers. In addition, we add two convolution layers with progressively decreasing dilation to the final group of the network, adopting filters of appropriate frequency, so that gridding artifacts are better eliminated. Compared to ResNet50 which has an output resolution of 7 &#x00D7; 7, the proposed dilated residual network with an output size of 28 &#x00D7; 28 can obtain more spatial information and identify small polyps without wasting unnecessary computational power to improve the classification performance of the network.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Attention Mechanism</title>
<p>With the attention mechanism, the deep learning network can divert attention to the most important areas of an image while ignoring irrelevant parts, which can be treated as a dynamic weight adjustment process according to the features of the input image. To bring performance gain further for our proposed network and avoid high model complexity, we apply ECA [<xref ref-type="bibr" rid="ref-26">26</xref>], a lightweight channel attention module, to all the residual blocks of IDE and ConvE in our network. The residual block structures of IDE Block and ConvE Block with the incorporation of the ECA module are shown in <xref ref-type="fig" rid="fig-2">Figs. 2a</xref> and <xref ref-type="fig" rid="fig-2">2b</xref>, respectively. As an improved attention mechanism based on Squeeze-and-Excitation Networks (SE) [<xref ref-type="bibr" rid="ref-27">27</xref>], ECA can effectively capture cross-channel interaction information without dimensionality reduction. The structure of the ECA module is presented in <xref ref-type="fig" rid="fig-2">Fig. 2c</xref>.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>The architectures of ECA and the residual blocks with ECA module in the proposed Polyp-DedNet. (a) ConvE block with the incorporation of ECA. (b) IDE block with the incorporation of ECA. (c) The structure of the ECA module</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_34720-fig-2.tif"/>
</fig>
<p>As illustrated in <xref ref-type="fig" rid="fig-2">Fig. 2c</xref>, after adopting channel-level GAP to aggregate features without dimensionality reduction, a 1D convolution is implemented to capture local cross-channel interactions between each channel and the corresponding <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:mi>k</mml:mi></mml:math></inline-formula> neighbors. Additionally, the 1D convolution with the kernel size of <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mi>k</mml:mi></mml:math></inline-formula> can be selected adaptively, determining coverage of local cross-channel interaction. The parameter <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:mi>k</mml:mi></mml:math></inline-formula> is determined as</p>

<p><disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x03C8;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>C</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo>|</mml:mo><mml:mfrac><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>C</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>&#x03B3;</mml:mi></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mi>b</mml:mi><mml:mi>&#x03B3;</mml:mi></mml:mfrac><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>d</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula>where <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mi>&#x03B3;</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mi>b</mml:mi></mml:math></inline-formula> are hyperparameters, C is the channel dimension, and <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>x</mml:mi><mml:msub><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>d</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the nearest odd number of <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>x</mml:mi></mml:math></inline-formula>.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>DropBlock</title>
<p>In this study, the collected polyp datasets are unbalanced, and the amount of image data in some categories is relatively small, which is prone to over-fitting in the training process of the proposed network. Therefore, we adopt a regularization method DropBlock after convolution layers (ConvE-layer) and residual modules (IDE block) in stages 4 and 5 of the proposed network to improve generalization ability and reduce overfitting, which is shown in <xref ref-type="fig" rid="fig-1">Fig. 1b</xref>.</p>
<p>DropBlock has two hyperparameters, <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:mi>&#x03B3;</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:mi>m</mml:mi></mml:math></inline-formula>, with the former controlling denoting the numbers of features to be dropped and the latter representing the size of the contiguous area to be dropped. During network training, the block size <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mi>m</mml:mi></mml:math></inline-formula> is fixed as 7, and <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:mi>&#x03B3;</mml:mi></mml:math></inline-formula> is calculated by the following formula:</p>
<p><disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mi>&#x03B3;</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>k</mml:mi><mml:mi>e</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mi>n</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mi>m</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x00D7;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>m</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:math></disp-formula>where <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:mi>k</mml:mi><mml:mi>e</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>b</mml:mi></mml:math></inline-formula> denotes the probability of keeping an activation unit in traditional dropout, <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mi>n</mml:mi></mml:math></inline-formula> represents the size of a feature map, and <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>m</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> denotes the size of the valid seed region. However, DropBlock with a fixed <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:mi>k</mml:mi><mml:mi>e</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>b</mml:mi></mml:math></inline-formula> has been found to be ineffective [<xref ref-type="bibr" rid="ref-15">15</xref>]. In this paper, a linear reduction of <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:mi>k</mml:mi><mml:mi>e</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>b</mml:mi></mml:math></inline-formula> parameter is adopted, that is, with the increase of training steps, <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mi>k</mml:mi><mml:mi>e</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>b</mml:mi></mml:math></inline-formula> gradually decreases from 1 to 0.9.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiment</title>
<sec id="s4_1">
<label>4.1</label>
<title>Dataset</title>
<p>Colonoscopy images and pathological information were collected retrospectively from the Affiliated Hospital of Hebei University, from June 1, 2016 to June 1, 2019. All the specimens were examined by certified pathologists and histologically confirmed. Patients who had been histologically confirmed with conventional adenoma (tubular, villioustublar, villous), sessile serrated polyp/adenoma (SSAP), hyperplastic polyp and normal (no polyps), were included in this study. Only the unamplified images observed in conventional white-light or NBI mode were selected. In this study, insufficiently insufflated colorectal images and unclear images with stool residue, halation, or bleeding were excluded. Finally, we collected 2132 images from 436 patients that contain six types of endoscopic colorectal disease images, which are shown in <xref ref-type="table" rid="table-2">Table 2</xref>.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>The information of the collected datasets included in this study</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Colorectal polyp type</th>
<th>Image number</th>
</tr>
</thead>
<tbody>
<tr>
<td>Hyperplastic polyp</td>
<td>234</td>
</tr>
<tr>
<td>Sessile serrated polyp/adenoma (SSAP)</td>
<td>122</td>
</tr>
<tr>
<td>Tubular adenoma</td>
<td>651</td>
</tr>
<tr>
<td>Villioustublar adenoma</td>
<td>618</td>
</tr>
<tr>
<td>Villous adenoma</td>
<td>51</td>
</tr>
<tr>
<td>Normal</td>
<td>456</td>
</tr>
<tr>
<td>Total</td>
<td>2132</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In order to build the classification models, the above datasets were classified by patients and randomly divided into the training set and test set in a ratio of 4:1. To develop our deep learning network more effectively, the training set was increased by a factor of 4 with data augmentation methods of horizontal flip, vertical flip, noise, and rotation to 6823 images. Considering that the collected colonoscopy images ranged from 424 &#x00D7; 368 to 1920 &#x00D7; 1080 pixels, they were uniformly resized to 224 &#x00D7; 224 pixels according to the network training requirements.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Network Training Details</title>
<p>The Polyp-DedNet was pre-trained using the ImageNet Large Scale Visual Recognition Challenge datasets, that is, using transfer learning to improve the classification performance of the network. The network parameters were optimized by the Adam algorithm with an initial learning rate of 0.0001. All programs were run on an Ubuntu 18.04.5 LTS PC with one RTX 2080Ti GPU and Intel (R) Core i7-7800X 3.5-GHz CPU.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Loss Function</title>
<p>Cross-entropy (CE) loss function is often used for deep learning network training. In this paper, to address the problem of training on imbalanced data, we adopted CB sigmoid cross-entropy loss by introducing a weighted factor that was inversely proportional to the effective number of samples. The CB sigmoid cross-entropy loss is formulated as:</p>
<p><disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:msub><mml:mrow><mml:mtext>CB</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="italic">s</mml:mi><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">g</mml:mi><mml:mi mathvariant="italic">m</mml:mi><mml:mi mathvariant="italic">o</mml:mi><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">d</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi mathvariant="bold-italic">z</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B2;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:msup><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:msubsup><mml:mrow><mml:mo>&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msubsup><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mo>&#x2212;</mml:mo><mml:msubsup><mml:mi>z</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>where <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:mi mathvariant="bold-italic">z</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>z</mml:mi><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>]</mml:mo></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> represents the predicted value for all classes from Polyp-DedNet, and <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:mi>C</mml:mi></mml:math></inline-formula> is the total number of polyp types. <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:mi>y</mml:mi></mml:math></inline-formula> is the sample whose category is labeled as <inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mi>y</mml:mi></mml:math></inline-formula>. <inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> indicates the effective sample growth rate, <inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:mi>&#x03B2;</mml:mi><mml:mspace width="thinmathspace" /><mml:mi>&#x03F5;</mml:mi><mml:mspace width="thinmathspace" /><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Performance Metrics</title>
<p>To evaluate the classification performance of the Polyp-DedNet network, we introduced four metrics including accuracy, recall, precision, and F1 Score [<xref ref-type="bibr" rid="ref-28">28</xref>], which can be calculated by the following formulas:</p>
<p><disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mrow><mml:mi mathvariant="italic">A</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">u</mml:mi><mml:mi mathvariant="italic">r</mml:mi><mml:mi mathvariant="italic">a</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">y</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<p><disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mrow><mml:mtext>Precision&#xA0;</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<p><disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mrow><mml:mi mathvariant="italic">R</mml:mi><mml:mi mathvariant="italic">e</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">a</mml:mi><mml:mi mathvariant="italic">l</mml:mi><mml:mi mathvariant="italic">l</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<p><disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mspace width="thinmathspace" /><mml:mrow><mml:mi mathvariant="italic">S</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">o</mml:mi><mml:mi mathvariant="italic">r</mml:mi><mml:mi mathvariant="italic">e</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mi mathvariant="italic">R</mml:mi><mml:mi mathvariant="italic">e</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">a</mml:mi><mml:mi mathvariant="italic">l</mml:mi><mml:mi mathvariant="italic">l</mml:mi></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mrow><mml:mi mathvariant="italic">P</mml:mi><mml:mi mathvariant="italic">r</mml:mi><mml:mi mathvariant="italic">e</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">s</mml:mi><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">o</mml:mi><mml:mi mathvariant="italic">n</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="italic">R</mml:mi><mml:mi mathvariant="italic">e</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">a</mml:mi><mml:mi mathvariant="italic">l</mml:mi><mml:mi mathvariant="italic">l</mml:mi></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mi mathvariant="italic">P</mml:mi><mml:mi mathvariant="italic">r</mml:mi><mml:mi mathvariant="italic">e</mml:mi><mml:mi mathvariant="italic">c</mml:mi><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">s</mml:mi><mml:mi mathvariant="italic">i</mml:mi><mml:mi mathvariant="italic">o</mml:mi><mml:mi mathvariant="italic">n</mml:mi></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula>where TP, FP, TN and FN denote the numbers of true positives, false positives, true negatives, and false negatives, respectively.</p>
<p>In this study, 5-fold cross-validation was used to analyze the network performance for four- and six-category classifications. The difference between the proposed model Polyp-DedNet and comparative models were calculated for each performance metric using the paired t-test. The <italic>p</italic>-value of &#x003C;0.05 was considered to indicate a statistically significant difference between groups.</p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Results and Discussions</title>
<sec id="s5_1">
<label>5.1</label>
<title>Evaluation of Loss Function During Training</title>
<p>To evaluate the effect of loss function on data imbalance during training, we compared two specific methods&#x2019; performance, including CE loss and CB sigmoid cross-entropy loss. As shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, while the train loss and the test loss gradually converged to 0.20 &#x00B1; 0.03 and 0.65 &#x00B1; 0.03 when training with the proposed Polyp-DedNet with CE loss, they gradually converged to 0.04 &#x00B1; 0.01 and 0.10 &#x00B1; 0.01 with CB sigmoid cross-entropy loss. In this case, our Polyp-DedNet had a faster convergence rate and a lower value for final loss with the implementation of CB sigmoid cross-entropy loss for network training compared with the use of CE loss. To a certain extent, the performance degradation caused by data imbalance was alleviated, especially with CB sigmoid cross-entropy loss.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>The dynamic changes of training loss and test loss during the training of Polyp-DedNet with CE loss and CB sigmoid cross-entropy loss</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_34720-fig-3.tif"/>
</fig>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Testing of the Proposed Method</title>
<p>The four- and six-category performance tests on a test set of 426 images were performed on the ResNet50 and Polyp-DedNet, respectively. Results of the five-fold cross-validation for the classifications were listed in <xref ref-type="table" rid="table-3">Table 3</xref>, in which the best for each metric was highlighted. For the four-category classification task, although Polyp-DedNet achieved insignificant increment in recall compared with the baseline ResNet50 (<italic>p</italic> &#x003D; 0.068), the remaining metrics including accuracy (89.91 &#x00B1; 0.92 <italic>vs</italic>. 88.46 &#x00B1; 0.79, <italic>p</italic> &#x003D; 0.001), precision (82.35 &#x00B1; 2.71 <italic>vs</italic>. 79.37 &#x00B1; 2.28, <italic>p</italic> &#x003D; 0.009) and F1-score (82.29 &#x00B1; 1.75 <italic>vs</italic>. 79.41 &#x00B1; 1.73, <italic>p</italic> &#x003D; 0.006) were significantly improved. Therefore, the overall performance of Polyp-DedNet for colorectal polyp classification was superior to that of ResNet50. For the six-category classification task, the cross-validation results illustrated the performance of the developed network was significantly improved in all evaluation metrics to varying degrees compared with the baseline network (<italic>p</italic> &#x003C; 0.05). The above results confirmed that extracting more spatial features from the input colonoscopy images can facilitate the network&#x2019;s effect on multi-classification.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Comparison of performances between ResNet50 and Polyp-DedNet tested with five-fold cross-validation. Significant differences according to the paired t test are shown in the last row</title>
</caption>
<table frame="hsides">
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Network</th>
<th colspan="2">Accuracy%</th>
<th colspan="2">Precision%</th>
<th colspan="2">Recall%</th>
<th colspan="2">F1-score%</th>
</tr>
<tr>
<td></td>
<td>Four-class</td>
<td>Six-class</td>
<td>Four-class</td>
<td>Six-class</td>
<td>Four-class</td>
<td>Six-class</td>
<td>Four-class</td>
<td>Six-class</td>
</tr>
</thead>
<tbody>
<tr>
<td>ResNet50</td>
<td>88.46 <inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 0.79</td>
<td>82.08 <inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 0.55</td>
<td>79.37 <inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 2.28</td>
<td>80.89 <inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 2.89</td>
<td>80.03 <inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 2.96</td>
<td>78.96 <inline-formula id="ieqn-54"><mml:math id="mml-ieqn-54"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 2.85</td>
<td>79.41 <inline-formula id="ieqn-55"><mml:math id="mml-ieqn-55"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 1.73</td>
<td>79.51 <inline-formula id="ieqn-56"><mml:math id="mml-ieqn-56"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 2.11</td>
</tr>
<tr>
<td>Polyp-DedNet</td>
<td><bold>89.91 </bold><inline-formula id="ieqn-57"><mml:math id="mml-ieqn-57"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>0.92</bold></td>
<td><bold>85.13 </bold><inline-formula id="ieqn-58"><mml:math id="mml-ieqn-58"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>1.10</bold></td>
<td><bold>82.35 </bold><inline-formula id="ieqn-59"><mml:math id="mml-ieqn-59"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>2.71</bold></td>
<td><bold>84.30 </bold><inline-formula id="ieqn-60"><mml:math id="mml-ieqn-60"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>2.22</bold></td>
<td><bold>82.95 </bold><inline-formula id="ieqn-61"><mml:math id="mml-ieqn-61"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>3.16</bold></td>
<td><bold>80.91 </bold><inline-formula id="ieqn-62"><mml:math id="mml-ieqn-62"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>3.48</bold></td>
<td><bold>82.29 </bold><inline-formula id="ieqn-63"><mml:math id="mml-ieqn-63"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>1.75</bold></td>
<td><bold>81.98 </bold><inline-formula id="ieqn-64"><mml:math id="mml-ieqn-64"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>2.60</bold></td>
</tr>
<tr>
<td>Significance</td>
<td><italic>p</italic> &#x003D; 0.001</td>
<td><italic>p</italic> &#x003D; 0.005</td>
<td><italic>p</italic> &#x003D; 0.009</td>
<td><italic>p</italic> &#x003D; 0.02</td>
<td>NS</td>
<td><italic>p</italic> &#x003D; 0.014</td>
<td><italic>p</italic> &#x003D; 0.006</td>
<td><italic>p</italic> &#x003D; 0.031</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5_3">
<label>5.3</label>
<title>Analysis of Confusion Matrix</title>
<p>In order to further compare the classification ability between ResNet50 and Polyp-DedNet, heat maps of the confusion matrix for the four- and six-category classifications were shown in <xref ref-type="fig" rid="fig-4">Fig. 4</xref>, in which the row and column represented the actual category and predicted category, respectively.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Confusion matrix of ResNet50 and Polyp-DedNet for classification of colorectal polyps on 426 images in our test set. (a) Matrix of ResNet50 for four classes. (b) Matrix of Polyp-DedNet for four classes. (c) Matrix of ResNet50 for six classes. (d) Matrix of Polyp-DedNet for six classes</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_34720-fig-4.tif"/>
</fig>
<p>In the four-classification task, the proposed Polyp-DedNet outperformed ResNet50 in classifying colorectal polyp images of adenomas and normal, with similar performance in classifying hyperplastic polyps and sessile serrated adenoma/polyp (SSAP). In addition, both networks provided more than 90% achievable performance for the classifications of adenomas and normal. In contrast, the performances of classifying other categories of polyps were relatively lower with the fact that most of them were misclassified as adenomas (see <xref ref-type="fig" rid="fig-4">Figs. 4a</xref> and <xref ref-type="fig" rid="fig-4">4b</xref>). It is worth mentioning that the training data for hyperplastic polyps and sessile serrated adenoma/polyp (SSAP) were much less than that of adenomas, and SSAP was more difficult to be distinguished than other polyps in clinical practice. In the six-category classification task, although our proposed framework provided comparable performance for the classifications of normal and villous adenomas with ResNet50, it achieved higher performance consistently for the classifications of hyperplasia, sessile serrated adenoma/polyp (SSAP), tubular adenoma and villoustubular adenoma with the improvements varied from 2% to 6%. In particular, the network Polyp-DedNet correctly classified more than 90% of normal and tubular adenoma images due to its increased focus on high-dimensional spatial information.</p>

</sec>
<sec id="s5_4">
<label>5.4</label>
<title>Analysis of ROC Curve</title>
<p>We further measured the standalone performance of the networks by the overall and class-specific ROC curves and the area under ROC curve (AUC) of four and six pathological types with an epoch set of 100, respectively. The larger the value of AUC, the better the performance of the network. The metric changes tend to level off after an average of 70 epochs. For the four-category classification, as shown in <xref ref-type="fig" rid="fig-5">Figs. 5a</xref> and <xref ref-type="fig" rid="fig-5">5b</xref>, the Polyp-DedNet was almost at a higher classification level in adenoma, normal and SSAP. The AUCs were as follows: adenoma, 0.9645 <italic>vs</italic>. 0.9742, normal, 0.9949 <italic>vs</italic>. 0.9966, SSAP, 0.9576 <italic>vs</italic>. 0.9587. For the classification of hyperplastic polyps, Polyp-DedNet performed slightly worse than ResNet50 (AUC, 0.9073 <italic>vs</italic>. 0.9062).</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Receiver Operating Characteristic (ROC) Curves for four and six classifications of ResNet50 and Polyp-DedNet in our test datasets (a) ROC of ResNet50 for four classes (b) ROC of Polyp-DedNet for four classes (c) ROC of ResNet50 for six classes (d) ROC of Polyp-DedNet for six classes</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_34720-fig-5.tif"/>
</fig>
<p>In the six-category classification, the AUCs of Polyp-DedNet were significantly higher than those of ResNet50 for the hyperplastic polyps, tubular, villioustublar and villious adenomas. However, the Polyp-DedNet performed slightly worse for normal and SSAP with lower AUCs than the pre-improved ResNet50. In addition, the improved network Polyp-DedNet showed overall competitive multi-classification performance with a mean AUC of 0.9593.</p>
</sec>
<sec id="s5_5">
<label>5.5</label>
<title>Analysis of Class Activation Map</title>
<p>To display the area of interest in the colonoscopy image more intuitively, representative class activation map (CAM) images of ResNet50 and Polyp-DedNet for multi-classification of polyps were generated, including hyperplastic polyps, (tubular, villioustublar and villious) adenomas and SSAP. As shown in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>, while both the networks ResNet50 and Polyp-DedNet could accurately localize hyperplastic polyps and villous adenomas (see the column (a) and (c) of <xref ref-type="fig" rid="fig-6">Fig. 6</xref>), our proposed network Polyp-DedNet captured villioustublar adenomas (see the column (b) of <xref ref-type="fig" rid="fig-6">Fig. 6</xref>) and tubular adenomas (see the column (e) of <xref ref-type="fig" rid="fig-6">Fig. 6</xref>) more accurately. It can be also observed that when the point of interest contained distinct categories of targets (i.e., a SSAP indicated by the red arrow and a non-SSAP polyp indicated by the yellow dashed box in the column (d) of <xref ref-type="fig" rid="fig-6">Fig. 6</xref>), the attended areas of ResNet50 were broadly distributed on both of them, whereas the attention focus of Polyp-DedNet was more precise on the characteristic area of SSAP, thereby extracting pathological features of colorectal polyps more effectively. Therefore, the extracted classification discrimination area of the proposed Polyp-DedNet was more consistent with the cognition of experts, which partially enhanced the applicability of deep learning.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Class activation maps of ResNet50 and Polyp-DedNet for classification of several colorectal polyps. (a) Hyperplastic polyp. (b) Villioustublar adenoma. (c) Villous adenoma. (d) SSAP. (e) Tubular adenoma. The red arrows highlight the locations of polyps</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_34720-fig-6.tif"/>
</fig>
</sec>
<sec id="s5_6">
<label>5.6</label>
<title>Comparative Analysis with Other Networks</title>
<p>In order to further verify the colorectal polyp classification performance of Polyp-DedNet, we performed comparative experiments with several state-of-the-art methods in terms of accuracy, precision, recall and F1-score, in which all the deep learning networks were built based on pre-trained ImageNet. In order to ensure the fairness of test results comparison, each network was trained under the same condition and tested on the same datasets. The six-category test results from five-fold cross-validation for each network were presented in <xref ref-type="table" rid="table-4">Table 4</xref>. It can be seen that Polyp-DedNet achieved the best performance in terms of selected evaluation indicators, which yielded an accuracy of 85.13% &#x00B1; 1.10%, a precision of 84.30% &#x00B1; 2.22%, a recall of 80.91% &#x00B1; 3.48% and an F1-score of 81.98% &#x00B1; 2.60%. As shown in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>, Polyp-DedNet was remarkably superior to MobileNetV3 according to all four indicators (<italic>p</italic> &#x003C; 0.01, paired t test). Besides, the precision of Polyp-DedNet was significantly higher than that of EfficientNetV2, and the metrics of recall and F1-score were consistently higher than those of RegNet (<italic>p</italic> &#x003C; 0.01). In addition, compared with EfficientNetV2, Polyp-DedNet enhanced in various indicators and achieved a statistically significant improvement in the average precision (1.64%; <italic>p</italic> &#x003C; 0.05) of multi-classification. In summary, with the more effective ability of feature extraction, the proposed Polyp-DedNet has excellent recognition ability for colorectal polyps with varying pathological types.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Comparative results of various polyp classification methods in our test datasets</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Network</th>
<th>Accuracy%</th>
<th>Precision%</th>
<th>Recall%</th>
<th>F1-score%</th>
</tr>
</thead>
<tbody>
<tr>
<td>MobileNetV3 [<xref ref-type="bibr" rid="ref-29">29</xref>]</td>
<td>77.06 <inline-formula id="ieqn-65"><mml:math id="mml-ieqn-65"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 1.85</td>
<td>76.41 <inline-formula id="ieqn-66"><mml:math id="mml-ieqn-66"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 1.79</td>
<td>67.22 <inline-formula id="ieqn-67"><mml:math id="mml-ieqn-67"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 3.06</td>
<td>70.16 <inline-formula id="ieqn-68"><mml:math id="mml-ieqn-68"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 2.74</td>
</tr>
<tr>
<td>RegNet [<xref ref-type="bibr" rid="ref-30">30</xref>]</td>
<td>82.88 <inline-formula id="ieqn-69"><mml:math id="mml-ieqn-69"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 1.78</td>
<td>82.93 <inline-formula id="ieqn-70"><mml:math id="mml-ieqn-70"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 1.18</td>
<td>76.17 <inline-formula id="ieqn-71"><mml:math id="mml-ieqn-71"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 3.07</td>
<td>78.68 <inline-formula id="ieqn-72"><mml:math id="mml-ieqn-72"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 2.19</td>
</tr>
<tr>
<td>EfficientNetV2 [<xref ref-type="bibr" rid="ref-31">31</xref>]</td>
<td>84.57 <inline-formula id="ieqn-73"><mml:math id="mml-ieqn-73"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 0.99</td>
<td>82.66 <inline-formula id="ieqn-74"><mml:math id="mml-ieqn-74"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 2.26</td>
<td>79.69 <inline-formula id="ieqn-75"><mml:math id="mml-ieqn-75"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 1.60</td>
<td>80.60 <inline-formula id="ieqn-76"><mml:math id="mml-ieqn-76"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> 0.83</td>
</tr>
<tr>
<td>Polyp-DedNet (Ours)</td>
<td><bold>85.13 </bold><inline-formula id="ieqn-77"><mml:math id="mml-ieqn-77"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>1.10</bold></td>
<td><bold>84.30 </bold><inline-formula id="ieqn-78"><mml:math id="mml-ieqn-78"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>2.22</bold></td>
<td><bold>80.91 </bold><inline-formula id="ieqn-79"><mml:math id="mml-ieqn-79"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>3.48</bold></td>
<td><bold>81.98 </bold><inline-formula id="ieqn-80"><mml:math id="mml-ieqn-80"><mml:mo>&#x00B1;</mml:mo></mml:math></inline-formula> <bold>2.60</bold></td>
</tr>
</tbody>
</table>
</table-wrap><fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Performance comparison between the Polyp-DedNet and contrast networks of MobileNet, RegNet and EffcientNetV2, in which &#x002A; denotes <italic>p</italic> &#x003C; 0.05, &#x002A;&#x002A; denotes <italic>p</italic> &#x003C; 0.01 and &#x002A;&#x002A;&#x002A; denotes <italic>p</italic> &#x003C; 0.001</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_34720-fig-7.tif"/>
</fig>
</sec>
</sec>
<sec id="s6">
<label>6</label>
<title>Conclusion</title>
<p>In this paper, we proposed a convolutional neural network named Polyp-DedNet, which is applied to the task of four- and six-category classifications of colorectal polyps based on the WASP and WHO medical classification standards under white light and narrow-band light. The experimental results of the four- and six-category classifications showed that the developed Polyp-DedNet could be used in the multi-classification task of common colorectal polyps, which helps to reduce superfluous resection and improve the sensitivity for the defection of early lesions in the clinical field. In the future, we will need to rationally validate the network through randomized clinical trials to help accurately classify colorectal polyps in actual colorectal examinations and assist doctors in choosing the best treatment strategy.</p>
</sec>
</body>
<back>
<sec><title>Funding Statement</title>
<p>This work was funded by the Research Fund for Foundation of Hebei University (DXK201914), the President of Hebei University (XZJJ201914), the Post-graduate&#x2019;s Innovation Fund Project of Hebei University (HBU2022SS003), the Special Project for Cultivating College Students&#x0027; Scientific and Technological Innovation Ability in Hebei Province (22E50041D), Guangdong Basic and Applied Basic Research Foundation (2021A1515011654), and the Fundamental Research Funds for the Central Universities of China (20720210117).</p>
</sec>
<sec sec-type="COI-statement"><title>Conflicts of Interest</title>
<p>We declare that we do not have any commercial or associative interest that represents a conflict on interest in connection with the work submitted.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Cao</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Da Chen</surname></string-name>, <string-name><given-names>Y. W.</given-names> <surname>Yu</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Li</surname></string-name> and <string-name><given-names>W. Q.</given-names> <surname>Chen</surname></string-name></person-group>, &#x201C;<article-title>Changing profiles of cancer burden worldwide and in China: A secondary analysis of the global cancer statistics 2020</article-title>,&#x201D; <source>Chinese Medical Journal</source>, vol. <volume>134</volume>, no. <issue>7</issue>, pp. <fpage>783</fpage>&#x2013;<lpage>791</lpage>, <year>2021</year>; <pub-id pub-id-type="pmid">33734139</pub-id></mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A. G.</given-names> <surname>Zauber</surname></string-name>, <string-name><given-names>S. J.</given-names> <surname>Winawer</surname></string-name>, <string-name><given-names>M. J.</given-names> <surname>O&#x2019;Brien</surname></string-name>, <string-name><given-names>I.</given-names> <surname>Lansdorp-Vogelaar</surname></string-name>, <string-name><given-names>M.</given-names> <surname>van Ballegooijen</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Colonoscopic polypectomy and long-term prevention of colorectal-cancer deaths</article-title>,&#x201D; <source>New England Journal of Medicine</source>, vol. <volume>366</volume>, no. <issue>8</issue>, pp. <fpage>687</fpage>&#x2013;<lpage>696</lpage>, <year>2012</year>; <pub-id pub-id-type="pmid">22356322</pub-id></mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D. K.</given-names> <surname>Rex</surname></string-name>, <string-name><given-names>C. R.</given-names> <surname>Boland</surname></string-name>, <string-name><given-names>J. A.</given-names> <surname>Dominitz</surname></string-name>, <string-name><given-names>F. M.</given-names> <surname>Giardiello</surname></string-name>, <string-name><given-names>D. A.</given-names> <surname>Johnson</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Colorectal cancer screening: Recommendations for physicians and patients from the U.S. multi-society task force on colorectal cancer</article-title>,&#x201D; <source>The American Journal of Gastroenterology</source>, vol. <volume>112</volume>, no. <issue>7</issue>, pp. <fpage>1016</fpage>&#x2013;<lpage>1030</lpage>, <year>2017</year>; <pub-id pub-id-type="pmid">28555630</pub-id></mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>B. I.</given-names> <surname>Lee</surname></string-name>, <string-name><given-names>S. P.</given-names> <surname>Hong</surname></string-name>, <string-name><given-names>S. E.</given-names> <surname>Kim</surname></string-name>, <string-name><given-names>S. H.</given-names> <surname>Kim</surname></string-name>, <string-name><given-names>H. S.</given-names> <surname>Kim</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Korean guidelines for colorectal cancer screening and polyp detection</article-title>,&#x201D; <source>Clinical Endoscopy</source>, vol. <volume>45</volume>, no. <issue>1</issue>, pp. <fpage>25</fpage>&#x2013;<lpage>43</lpage>, <year>2012</year>; <pub-id pub-id-type="pmid">22741131</pub-id></mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W. B.</given-names> <surname>Strum</surname></string-name></person-group>, &#x201C;<article-title>Colorectal adenomas</article-title>,&#x201D; <source>New England Journal of Medicine</source>, vol. <volume>374</volume>, no. <issue>11</issue>, pp. <fpage>1065</fpage>&#x2013;<lpage>1075</lpage>, <year>2016</year>; <pub-id pub-id-type="pmid">26981936</pub-id></mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Bibbins-Domingo</surname></string-name>, <string-name><given-names>D. C.</given-names> <surname>Grossman</surname></string-name>, <string-name><given-names>S. J.</given-names> <surname>Curry</surname></string-name>, <string-name><given-names>K. W.</given-names> <surname>Davidson</surname></string-name>, <string-name><given-names>J. W.</given-names> <surname>Epling</surname> <suffix>Jr</suffix></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Screening for colorectal cancer: US preventive services task force recommendation statement</article-title>,&#x201D; <source>JAMA</source>, vol. <volume>315</volume>, no. <issue>23</issue>, pp. <fpage>2564</fpage>&#x2013;<lpage>2575</lpage>, <year>2016</year>; <pub-id pub-id-type="pmid">27304597</pub-id></mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. A.</given-names> <surname>Almadi</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Sewitch</surname></string-name>, <string-name><given-names>A. N.</given-names> <surname>Barkun</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Martel</surname></string-name> and <string-name><given-names>L.</given-names> <surname>Joseph</surname></string-name></person-group>, &#x201C;<article-title>Adenoma detection rates decline with increasing procedural hours in an endoscopist&#x2019;s workload</article-title>,&#x201D; <source>Canadian Journal of Gastroenterology &#x003D; Journal Canadien de Gastroenterologie</source>, vol. <volume>29</volume>, no. <issue>6</issue>, pp. <fpage>304</fpage>&#x2013;<lpage>308</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Yamada</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Sakamoto</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Otake</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Nakajima</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Kuchiba</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Investigating endoscopic features of sessile serrated adenomas/polyps by using narrow-band imaging with optical magnification</article-title>,&#x201D; <source>Gastrointestinal Endoscopy</source>, vol. <volume>82</volume>, no. <issue>1</issue>, pp. <fpage>108</fpage>&#x2013;<lpage>117</lpage>, <year>2015</year>; <pub-id pub-id-type="pmid">25840928</pub-id></mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. L.</given-names> <surname>Giger</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Doi</surname></string-name> and <string-name><given-names>H.</given-names> <surname>MacMahon</surname></string-name></person-group>, &#x201C;<article-title>Image feature analysis and computer-aided diagnosis in digital radiography. 3. Automated detection of nodules in peripheral lung fields</article-title>,&#x201D; <source>Medical Physics</source>, vol. <volume>15</volume>, no. <issue>2</issue>, pp. <fpage>158</fpage>&#x2013;<lpage>166</lpage>, <year>1988</year>; <pub-id pub-id-type="pmid">3386584</pub-id></mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Fan</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Xu</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Fan</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Wei</surname></string-name> and <string-name><given-names>L.</given-names> <surname>Li</surname></string-name></person-group>, &#x201C;<article-title>Computer-aided detection of small intestinal ulcer and erosion in wireless capsule endoscopy images</article-title>,&#x201D; <source>Physics in Medicine and Biology</source>, vol. <volume>63</volume>, no. <issue>16</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>10</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. A.</given-names> <surname>Khan</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Akram</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Sharif</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Javed</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Rashid</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>An integrated framework of skin lesion detection and recognition through saliency method and optimal deep neural network features selection</article-title>,&#x201D; <source>Neural Computing and Applications</source>, vol. <volume>32</volume>, no. <issue>20</issue>, pp. <fpage>15929</fpage>&#x2013;<lpage>15948</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J. E. G.</given-names> <surname>IJspeert</surname></string-name>, <string-name><given-names>B. A.</given-names> <surname>Bastiaansen</surname></string-name>, <string-name><given-names>M. E.</given-names> <surname>van Leerdam</surname></string-name>, <string-name><given-names>G. A.</given-names> <surname>Meijer</surname></string-name>, <string-name><given-names>S.</given-names> <surname>van Eeden</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Development and validation of the WASP classification system for optical diagnosis of adenomas, hyperplastic polyps and sessile serrated adenomas/polyps</article-title>,&#x201D; <source>Gut</source>, vol. <volume>65</volume>, no. <issue>6</issue>, pp. <fpage>963</fpage>&#x2013;<lpage>970</lpage>, <year>2016</year>; <pub-id pub-id-type="pmid">25753029</pub-id></mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C. C.</given-names> <surname>G&#x00F6;ret</surname></string-name> and <string-name><given-names>N. E.</given-names> <surname>G&#x00F6;ret</surname></string-name></person-group>, &#x201C;<article-title>Histopathological analysis of 173 consecutive patients with colorectal carcinoma: A pathologist&#x2019;s view</article-title>,&#x201D; <source>Medical Science Monitor: International Medical Journal of Experimental and Clinical Research</source>, vol. <volume>24</volume>, pp. <fpage>6809</fpage>&#x2013;<lpage>6815</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J. E. G.</given-names> <surname>Ijspeert</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Bevan</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Senore</surname></string-name>, <string-name><given-names>M. F.</given-names> <surname>Kaminski</surname></string-name>, <string-name><given-names>E. J.</given-names> <surname>Kuipers</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Detection rate of serrated polyps and serrated polyposis syndrome in colorectal cancer screening cohorts: A European overview</article-title>,&#x201D; <source>Gut</source>, vol. <volume>66</volume>, no. <issue>7</issue>, pp. <fpage>1225</fpage>&#x2013;<lpage>1232</lpage>, <year>2017</year>; <pub-id pub-id-type="pmid">26911398</pub-id></mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Ghiasi</surname></string-name>, <string-name><given-names>T. Y.</given-names> <surname>Lin</surname></string-name> and <string-name><given-names>Q. V.</given-names> <surname>Le</surname></string-name></person-group>, &#x201C;<article-title>DropBlock: A regularization method for convolutional networks</article-title>,&#x201D; <source>Advances in Neural Information Processing Systems</source>, vol. <volume>2018-December</volume>, pp. <fpage>10727</fpage>&#x2013;<lpage>10737</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Cui</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Jia</surname></string-name>, <string-name><given-names>T. -Y.</given-names> <surname>Lin</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Song</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Belongie</surname></string-name></person-group>, &#x201C;<article-title>Class-balanced loss based on effective number of samples</article-title>,&#x201D; in <conf-name>Proc. of the IEEE Conf. on Computer Vision and Pattern Recognition</conf-name>, <publisher-loc>Long Beach, CA, USA</publisher-loc>, pp. <fpage>9260</fpage>&#x2013;<lpage>9269</lpage>, <year>2019</year>. </mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Shin</surname></string-name> and <string-name><given-names>I.</given-names> <surname>Balasingham</surname></string-name></person-group>, &#x201C;<article-title>Automatic polyp frame screening using patch based combined feature and dictionary learning</article-title>,&#x201D; <source>Computerized Medical Imaging and Graphics</source>, vol. <volume>69</volume>, pp. <fpage>33</fpage>&#x2013;<lpage>42</lpage>, <year>2018</year>; <pub-id pub-id-type="pmid">30172091</pub-id></mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Tamaki</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Yoshimuta</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Takeda</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Raytchev</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Kaneda</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>A system for colorectal tumor classification in magnifying endoscopic NBI images</article-title>,&#x201D; <source>Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)</source>, vol. <volume>6493 LNCS</volume>, no. <issue>PART 2</issue>, pp. <fpage>452</fpage>&#x2013;<lpage>463</lpage>, <year>2011</year>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Poudel</surname></string-name>, <string-name><given-names>Y. J.</given-names> <surname>Kim</surname></string-name>, <string-name><given-names>D. M.</given-names> <surname>Vo</surname></string-name> and <string-name><given-names>S. W.</given-names> <surname>Lee</surname></string-name></person-group>, &#x201C;<article-title>Colorectal disease classification using efficiently scaled dilation in convolutional neural network</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>8</volume>, pp. <fpage>99227</fpage>&#x2013;<lpage>99238</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Tian</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Luo</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Wang</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>An improved deep learning approach and its applications on colonic polyp images detection</article-title>,&#x201D; <source>BMC Medical Imaging</source>, vol. <volume>20</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>15</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>P. J.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>M. C.</given-names> <surname>Lin</surname></string-name>, <string-name><given-names>M. J.</given-names> <surname>Lai</surname></string-name>, <string-name><given-names>J. C.</given-names> <surname>Lin</surname></string-name>, <string-name><given-names>H. H. S.</given-names> <surname>Lu</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Accurate classification of diminutive colorectal polyps using computer-aided analysis</article-title>,&#x201D; <source>Gastroenterology</source>, vol. <volume>154</volume>, no. <issue>3</issue>, pp. <fpage>568</fpage>&#x2013;<lpage>575</lpage>, <year>2018</year>; <pub-id pub-id-type="pmid">29042219</pub-id></mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. F.</given-names> <surname>Byrne</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Chapados</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Soudan</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Oertel</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Linares P&#x00E9;rez</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Real-time differentiation of adenomatous and hyperplastic diminutive colorectal polyps during analysis of unaltered videos of standard colonoscopy using a deep learning model</article-title>,&#x201D; <source>Gut</source>, vol. <volume>68</volume>, no. <issue>1</issue>, pp. <fpage>94</fpage>&#x2013;<lpage>100</lpage>, <year>2019</year>; <pub-id pub-id-type="pmid">29066576</pub-id></mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Ozawa</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ishihara</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Fujishiro</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Kumagai</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Shichijo</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Automated endoscopic detection and classification of colorectal polyps using convolutional neural networks</article-title>,&#x201D; <source>Therapeutic Advances in Gastroenterology</source>, vol. <volume>13</volume>, pp. <fpage>1</fpage>&#x2013;<lpage>13</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Feng</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Song</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Liu</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Liu</surname></string-name></person-group>, &#x201C;<article-title>Multiclassification of endoscopic colonoscopy images based on deep transfer learning</article-title>,&#x201D; <source>Computational and Mathematical Methods in Medicine</source>, vol. <volume>2021</volume>, pp. <fpage>1</fpage>&#x2013;<lpage>12</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>He</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ren</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Sun</surname></string-name></person-group>, &#x201C;<article-title>Deep residual learning for image recognition</article-title>,&#x201D; in <conf-name>Proc. of the IEEE Conf. on Computer Vision and Pattern Recognition</conf-name>, <publisher-loc>Las Vegas, NV, USA</publisher-loc>, pp. <fpage>770</fpage>&#x2013;<lpage>778</lpage>, <year>2016</year>. </mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Q.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Wu</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Zhu</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Zuo</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>ECA-Net: Efficient channel attention for deep convolutional neural networks</article-title>,&#x201D; in <conf-name>Proc. of the IEEE Conf. on Computer Vision and Pattern Recognition</conf-name>, <publisher-loc>Seattle, WA, USA</publisher-loc>, pp. <fpage>11531</fpage>&#x2013;<lpage>11539</lpage>, <year>2020</year>. </mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Hu</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Shen</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Albanie</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Sun</surname></string-name> and <string-name><given-names>E.</given-names> <surname>Wu</surname></string-name></person-group>, &#x201C;<article-title>Squeeze-and-excitation networks</article-title>,&#x201D; <source>IEEE Transactions on Pattern Analysis and Machine Intelligence</source>, vol. <volume>42</volume>, no. <issue>8</issue>, pp. <fpage>2011</fpage>&#x2013;<lpage>2023</lpage>, <year>2020</year>; <pub-id pub-id-type="pmid">31034408</pub-id></mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D. M. W.</given-names> <surname>Powers</surname></string-name></person-group>, &#x201C;<article-title>Evaluation: From precision, recall and F-measure to ROC, informedness, markedness and correlation</article-title>,&#x201D; <source>Journal of Machine Learning Technologies</source>, vol. <volume>2</volume>, no. <issue>1</issue>, pp. <fpage>37</fpage>&#x2013;<lpage>63</lpage>, <year>2011</year>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Howard</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Sandler</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>L. C.</given-names> <surname>Chen</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Searching for MobileNetV3</article-title>,&#x201D; in <conf-name>Proc. of the IEEE Conf. on Computer Vision and Pattern Recognition</conf-name>, <publisher-loc>Long Beach, CA, USA</publisher-loc>, pp. <fpage>1314</fpage>&#x2013;<lpage>1324</lpage>, <year>2019</year>. </mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>I.</given-names> <surname>Radosavovic</surname></string-name>, <string-name><given-names>R. P.</given-names> <surname>Kosaraju</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Girshick</surname></string-name>, <string-name><given-names>K.</given-names> <surname>He</surname></string-name> and <string-name><given-names>P.</given-names> <surname>Doll&#x00E1;r</surname></string-name></person-group>, &#x201C;<article-title>Designing network design spaces</article-title>,&#x201D; in <conf-name>Proc. of the IEEE Conf. on Computer Vision and Pattern Recognition</conf-name>, <publisher-loc>Seattle, WA, USA</publisher-loc>, pp. <fpage>10425</fpage>&#x2013;<lpage>10433</lpage>, <year>2020</year>. </mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Tan</surname></string-name> and <string-name><given-names>Q. V.</given-names> <surname>Le</surname></string-name></person-group>, &#x201C;<article-title>EfficientNetV2: Smaller models and faster training</article-title>,&#x201D; in <conf-name>Proc. of the 38th Int. Conf. on Machine Learning</conf-name>, vol. <volume>139</volume>, pp. <fpage>10096</fpage>&#x2013;<lpage>10106</lpage>, <year>2021</year>. </mixed-citation></ref>
</ref-list>
</back>
</article>