<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">16341</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2021.016341</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Robust Magnification Independent Colon Biopsy Grading System over Multiple Data Sources</article-title>
<alt-title alt-title-type="left-running-head">Robust Magnification Independent Colon Biopsy Grading System over Multiple Data Sources</alt-title>
<alt-title alt-title-type="right-running-head">Robust Magnification Independent Colon Biopsy Grading System over Multiple Data Sources</alt-title>
</title-group>
<contrib-group content-type="authors">
<contrib id="author-1" contrib-type="author">
<name name-style="western">
<surname>Babu</surname>
<given-names>Tina</given-names>
</name>
<xref ref-type="aff" rid="aff-1">1</xref>
</contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western">
<surname>Gupta</surname>
<given-names>Deepa</given-names>
</name>
<xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-3" contrib-type="author" corresp="yes">
<name name-style="western">
<surname>Singh</surname>
<given-names>Tripty</given-names>
</name>
<xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western">
<surname>Hameed</surname>
<given-names>Shahin</given-names>
</name>
<xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western">
<surname>Zakariah</surname>
<given-names>Mohammed</given-names>
</name>
<xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western">
<surname>Alotaibi</surname>
<given-names>Yousef Ajami</given-names>
</name>
<xref ref-type="aff" rid="aff-4">4</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Department of Computer Science and Engineering, Amrita School of Engineering, Amrita Vishwa Vidyapeetham</institution>, <addr-line>Bengaluru</addr-line>, <country>India</country></aff>
<aff id="aff-2"><label>2</label><institution>Department of Pathology, MVR Cancer Center and Research Institute</institution>, <addr-line>Poolacode, Kerala</addr-line>, <country>India</country></aff>
<aff id="aff-3"><label>3</label><institution>College of Computer and Information Sciences, King Saud University</institution>, <country>Saudi Arabia</country></aff>
<aff id="aff-4"><label>4</label><institution>Computer Engineering Department, College of Computer and Information Sciences, King Saud University</institution>, <country>Saudi Arabia</country></aff>
</contrib-group>
<author-notes><corresp id="cor1">&#x002A;Corresponding Author: Tripty Singh. Email: <email>tripty_singh@blr.amrita.edu</email></corresp></author-notes>
<pub-date pub-type="epub" date-type="pub" iso-8601-date="2021-05-31"><day>31</day><month>05</month><year>2021</year></pub-date>
<volume>69</volume>
<issue>1</issue>
<fpage>99</fpage>
<lpage>128</lpage>
<history>
<date date-type="received"><day>30</day><month>12</month><year>2020</year></date>
<date date-type="accepted"><day>13</day><month>03</month><year>2021</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2021 Babu et al.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Babu et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_16341.pdf"></self-uri>
<abstract>
<p>Automated grading of colon biopsy images across all magnifications is challenging because of tailored segmentation and dependent features on each magnification. This work presents a novel approach of robust magnification-independent colon cancer grading framework to distinguish colon biopsy images into four classes: normal, well, moderate, and poor. The contribution of this research is to develop a magnification invariant hybrid feature set comprising cartoon feature, Gabor wavelet, wavelet moments, HSV histogram, color auto-correlogram, color moments, and morphological features that can be used to characterize different grades. Besides, the classifier is modeled as a multiclass structure with six binary class Bayesian optimized random forest (BO-RF) classifiers. This study uses four datasets (two collected from Indian hospitals&#x2014;Ishita Pathology Center (IPC) of 4X, 10X, and 40X and Aster Medcity (AMC) of 10X, 20X, and 40X&#x2014;two benchmark datasets&#x2014;gland segmentation (GlaS) of 20X and IMEDIATREAT of 10X) comprising multiple microscope magnifications. Experimental results demonstrate that the proposed method outperforms the other methods used for colon cancer grading in terms of accuracy (97.25%-IPC, 94.40%-AMC, 97.58%-GlaS, 99.16%-Imediatreat), sensitivity (0.9725-IPC, 0.9440-AMC, 0.9807-GlaS, 0.9923-Imediatreat), specificity (0.9908-IPC, 0.9813-AMC, 0.9907-GlaS, 0.9971-Imediatreat) and F-score (0.9725-IPC, 0.9441-AMC, 0.9780-GlaS, 0.9923-Imediatreat). The generalizability of the model to any magnified input image is validated by training in one dataset and testing in another dataset, highlighting strong concordance in multiclass classification and evidencing its effective use in the first level of automatic biopsy grading and second opinion.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Colon cancer</kwd>
<kwd>grading</kwd>
<kwd>texture features</kwd>
<kwd>color features</kwd>
<kwd>morphological features</kwd>
<kwd>feature extraction</kwd>
<kwd>Bayesian optimized random forest classifier</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Colorectal cancer is one of the world&#x2019;s most common cancers and is the second leading cause of cancer death [<xref ref-type="bibr" rid="ref-1">1</xref>]. In 2018, it ranked the third and second-most-common cancer for both genders&#x2019; incidence and mortality globally, constituting respectively 6.1% and 5.8% of the number of new cases and deaths, among all cancers combined worldwide [<xref ref-type="bibr" rid="ref-2">2</xref>]. The general cancer diagnosis process is tedious and reliant on experts using microscopic analysis of biopsy samples. An essential task for pathologists who analyze colon specimens across various magnifications in a microscope (4X, 5X, 10X, 20X, and 40X) is to distinguish invasive cancer and, to provide an accurate diagnosis and grading critical for the treatment plan. The subjective character of grading evaluation and the different patterns that many tumors exhibit render it difficult to achieve consistency between pathologists. This method requires a substantial amount of time to provide results in both inter-and intra-observer variations [<xref ref-type="bibr" rid="ref-3">3</xref>,<xref ref-type="bibr" rid="ref-4">4</xref>]. Owing to the visual discrepancy among observations, analyzing the sample under a microscope at various magnifications is crucial for an accurate diagnosis. The golden standard for diagnosis is an analysis by pathologists with subspecific expertise and specialty in gastrointestinal malignancy. However, second opinions are slow to come, work-intensive, and often not possible in areas with scarce resources. Advanced computerized pathology over numerous magnifications offers an assisted and suitable solution to this issue [<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-5">5</xref>]. In particular, with numerous digitized images of histology slides being progressively ubiquitous, automated diagnosis can help the pathologist by providing second opinions through machine learning. Automatic cancer screening is the first level of diagnosis followed by grades determination across various magnifications. To solve this multiclass classification problem, a magnification-independent framework is essential for investigating pathological images using image processing and machine learning techniques.</p>
<p>Most medical applications use image features and image processing techniques [<xref ref-type="bibr" rid="ref-6">6</xref>]. A very recent and comprehensive literature review was performed to extract clinical details from histological slides [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-8">8</xref>]. An overview of recent literature in two key directions on colon cancer diagnosis, i.e., detection and grading of colon biopsy images, is reviewed in the current research.</p>
<p>Several automated approaches are available to distinguish between normal and malignant colon lesions. Rathore et al. [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-10">10</xref>] proposed an ellipse fitting algorithm with K-means clustering to segment the glands specifically on 10X magnified colon images and extracted a hybrid feature set (morphological, geometric, texture-based, scale-invariant feature transform, and elliptical Fourier descriptor features) and lumen characteristic dependent on the segmented region of interest (ROI)s and classified with SVM classifier into normal and malignant images. Furthermore, Rathore et al. [<xref ref-type="bibr" rid="ref-11">11</xref>] optimized the segmentation parameters for each magnification (4X, 5X, 10X, and 40X) for ellipse fitting algorithm using genetic algorithm and extracted gray-level co-occurrence matrix (GLCM)-based as well as gray-level histogram moment features from the segmented ROI to classify colon biopsy images through an SVM classifier, thereby attaining 92.33% average accuracy. Across various magnified colon images (10X, 20X, 40X), for cancer detection, texture, shape, and wavelet features were analyzed and classified using multi-classifier models in [<xref ref-type="bibr" rid="ref-12">12</xref>&#x2013;<xref ref-type="bibr" rid="ref-15">15</xref>]. Abdulhay et al. [<xref ref-type="bibr" rid="ref-16">16</xref>] suggested a strategy for the segmentation of blood leukocytes using static microscopes to classify 100 unique magnified microscopic pictures (72-abnormal, 38-normal) by using SVM for the tuned segmentation and filtering of the non-ROI image using local binary patterns and texture characteristics with a 95.3% accuracy. With image, local, and gland features extracted from image-specific tuned the multistep gland segmentation, Rathore et al. [<xref ref-type="bibr" rid="ref-17">17</xref>] encoded the glandular patterns and morphology of cells and detected cancer using a score-based ensemble SVM classifier. Their method was evaluated on the GlaS dataset [<xref ref-type="bibr" rid="ref-18">18</xref>] and 10X-magnified colon biopsy images, attaining accuracies of 98.30% and 97.60%, respectively. For 100 samples of BRATS Brain MRI data sets, Husham et al. [<xref ref-type="bibr" rid="ref-19">19</xref>] compared active contour and otsu threshold algorithms where the segmentation parameters were set for that dataset, and the supremacy of active contour was confirmed. Hussein et al. [<xref ref-type="bibr" rid="ref-20">20</xref>] proposed a new version for Viola-James that segments ultrasound images of the breast (250 images) and ovarian (100 images) that generate ROI with active contour tuned for these images and magnification and achieved a classification accuracy of 95.43% and 94.84.0% for breast and ovarian images with new features dependent on the segmented region, characterize the lesion. Recently, deep neural networks have been widely applied in medical image processing and digital pathology [<xref ref-type="bibr" rid="ref-21">21</xref>]. Motivated by the LeNet-5 structure, glandular artifact and clustered gland segregation were detected using two convolutional neural networks (CNN) [<xref ref-type="bibr" rid="ref-22">22</xref>]. Further, cancer was detected with 95% accuracy using the 20X-magnified images of the GlaS dataset. Xu et al. [<xref ref-type="bibr" rid="ref-23">23</xref>] utilized the activation features extracted from the CNN trained on Imagenet for segmentation and classification. The SVM classifier was used to classify the 10X-magnified colon and brain biopsy images with 98% and 97.8% accuracy, respectively. A deep CNN network was used for gland segmentation and characterization; then, the best alignment matrix (BAM) feature extracted from this segmented region was used for two-class classification with a 97% accuracy on the GlaS dataset [<xref ref-type="bibr" rid="ref-24">24</xref>]. Later, Lichtblau et al. [<xref ref-type="bibr" rid="ref-25">25</xref>] implemented transfer learning on Alexnet to extract high-level features to classify the target images into benign and malignant samples with six classifiers&#x2019; probability score. The classifier weights are optimized via differential evolution and achieved an accuracy of 96.66% on the GlaS dataset, and with BreaKHis [<xref ref-type="bibr" rid="ref-26">26</xref>] dataset accuracies of 83.9%, 86%, 89.1%, and 86.6% were tabulated for 40X, 100X, 200X, and 400X magnified microscopic images respectively. Iizuka et al. [<xref ref-type="bibr" rid="ref-27">27</xref>] extracted the high-level features with the Inception-v3 CNN network. They used a recurrent neural network and max-pooling to classify the images into two classes: adenocarcinoma, adenoma of the stomach, and colon whole slide images with an area under the curve of 0.980, 0.974, respectively.</p>
<p>Many techniques have been explored in the grading/multiclass classification of colon biopsy images. Rathore et al. [<xref ref-type="bibr" rid="ref-10">10</xref>], using 10X magnified colon images, graded the malignant images into three classes: well, moderate, and poor with an SVM classifier based on the lumen area characteristics extracted from the lumen through the ellipse fitting algorithm on the white cluster obtained through K-means clustering for this dataset with 93.47% accuracy. Furthermore, Kather et al. [<xref ref-type="bibr" rid="ref-28">28</xref>], using conventional features such as GLCM, Histogram, local binary patterns, and Gabor, classified the colon tissue samples into eight classes utilizing an SVM classifier with 87.4% accuracy. With the GlaS dataset, Saroja et al. [<xref ref-type="bibr" rid="ref-29">29</xref>] implemented adaptive pillar K-means clustering to extract the lumen features; then, using a score-based decision tree, graded the malignant colon images into three classes with 93% accuracy. Boruz et al. [<xref ref-type="bibr" rid="ref-30">30</xref>], based on the texture and topological features extracted from the gland segmented image, classified the 10X-magnified Imediatreat [<xref ref-type="bibr" rid="ref-31">31</xref>] colon image dataset into four classes: healthy, well, moderate, and poor, and obtained an accuracy of 89.75% with an SVM classifier. The cell morphology, glandular structures, and texture are considered from tailored multi-step gland segmentation for the 10X-magnified images and GlaS dataset. The image, local, and gland features are extracted from these segmented images and graded malignant colon images into three classes; therein, both datasets achieved 98.6% accuracy using score-based ensemble SVM [<xref ref-type="bibr" rid="ref-17">17</xref>]. Nawadhar et al. [<xref ref-type="bibr" rid="ref-32">32</xref>] proposed a stratified squamous epithelial biopsy image classifier that takes majority voting of the five classifiers for grading 676 oral mucosa 40X-magnified images into four classes: normal, well, moderate, and poor with 95.56% accuracy with the color, texture and shape features extracted from the segmented region. The cellular regions were segmented with unsupervised K-means clustering and Moore-neighbor tracing algorithm with Jacob&#x2019;s stopping criteria tuned for this dataset. Rathore et al. [<xref ref-type="bibr" rid="ref-33">33</xref>], with ROI, delineated 20X glioma images, graded into high and low grades with the conventional, clinical, and texture features dependent on the ROI, with SVM classifier with 91.48% accuracy. Deep learning techniques were also explored for the grading or multiclass classification of biopsy images. Xu et al. [<xref ref-type="bibr" rid="ref-23">23</xref>], with the high-level features extracted from the Imagenet CNN model, segmentation of patches is performed with supervised learning using linear SVM and classified the 10X-magnified colon tissue images into six classes with 87% accuracy. Gland segmentation was performed using CNN based on UNet architecture, wherein BAM was extracted from the segmented glands, thereby using glandular aberration features with the SVM classifier for grading 20X-magnified colon biopsy images into three classes: normal, low grade, and high grade with 95.33% accuracy [<xref ref-type="bibr" rid="ref-24">24</xref>]. Lichtblau et al. [<xref ref-type="bibr" rid="ref-25">25</xref>] optimized the ensemble weights of six distinct classifiers with differential evolution algorithm, thereby considering individual classifiers probabilities for grading each sample into four classes. Thereby, using to grade 10X-magnified colon image Imediatreat [<xref ref-type="bibr" rid="ref-31">31</xref>] dataset into four classes using the activation features extracted from the Alexnet CNN model with 98.29% accuracy.</p>
<p>In the majority of literature, where color-based clustering, segmentation, and features [<xref ref-type="bibr" rid="ref-9">9</xref> &#x2013;<xref ref-type="bibr" rid="ref-11">11</xref>] are used, the techniques depend on the image color intensities that subsequently depend on the staining concentrations and illumination conditions [<xref ref-type="bibr" rid="ref-5">5</xref>]; hence, affect the post-processing using color features [<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>]. Besides, the traditional approaches for cancer detection or grading include segmentation methods tuned for specific magnified images (mostly 10X-magnified) and performance deteriorates with other image magnifications (4X, 20X, 40X) as the parameters are set for a particular magnification [<xref ref-type="bibr" rid="ref-9">9</xref>&#x2013;<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-16">16</xref>,<xref ref-type="bibr" rid="ref-17">17</xref>,<xref ref-type="bibr" rid="ref-19">19</xref>,<xref ref-type="bibr" rid="ref-20">20</xref>,<xref ref-type="bibr" rid="ref-29">29</xref>,<xref ref-type="bibr" rid="ref-30">30</xref>]. Thus, finding a region of interest (ROI) is tedious for each image magnification. Further, features extracted from these segmented regions, including geometric, lumen, morphological, and topological features that depend on spatial domain, differ across image magnification. Although deep learning plays a vital role in many classification problems where CNN automatically and optimally adjusts feature extraction for the desired classification [<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-27">27</xref>], it requires massive, detailed annotated medical data that is scarce, complex hardware, and high computation time. Binary class problems are better classified using deep learning models. However, for grading or multiclass problems, activation features are extracted from the existing CNN models, and classifiers are optimized to boost classification accuracy [<xref ref-type="bibr" rid="ref-23">23</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>]. Moreover, in traditional methods and deep learning models, training and testing were performed with the respective datasets and magnification. A thorough literature review reveals the need for an efficient magnification-independent colon cancer grading framework for biopsy images applicable across various H&#x0026;E colon biopsy image datasets.</p>
<p>This work&#x2019;s primary objective is to simplify the automated magnification-independent four-class grading framework on a set of images from histopathological colon tissue slides where the grading ranges from normal/healthy to three grade levels&#x2014;well, moderate, and poor. A robust magnification-invariant rich hybrid feature set is proposed that explores the structural, textural, color, and shape properties across magnifications. Further, training ensembles of Bayesian optimized random forest classifiers eased the grading problem by using a majority voting to obtain the final classification label. The pursued contributions are as follows.</p>
<list list-type="bullet">
<list-item><p>Image pre-processing as stain normalization for stain concentrations to ensure image uniformity within and across multiple datasets.</p></list-item>
<list-item><p>A robust, rich hybrid feature set independent of the spatial variations is proposed, containing texture (cartoon features, Gabor wavelet, wavelet moments), color (HSV histogram, color auto-correlogram, color moments), and morphological features.</p></list-item>
<list-item><p>Using Ensemble Bayesian Optimized Random Forest classifiers, the proposed framework classifies the images as a multiclass structure with six classifiers to ease the multiclass grading problem, and according to the maximum similar population, the final class is predicted to ensure optimal classification accuracy.</p></list-item>
<list-item><p>The model&#x2019;s generalizability proposed on various magnified datasets is evaluated across four colon biopsy image datasets (two collected from Indian hospitals and two benchmark datasets).</p></list-item>
<list-item><p>Training in one dataset and testing with other datasets provides a better outcome with the robust classification model, ensuring any magnified input colon biopsy images&#x2019; applicability.</p></list-item>
</list>
<p>The rest of the paper is organized as follows: Section 2 presents input colon image characteristics and datasets used, while the proposed methodology is described in Section 3. Performance measures used for evaluation and results are described in Section 4 and discussed in Section 5. Finally, in Section 6, the conclusion and future work are presented.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Input Colon Biopsy Image Datasets</title>
<p>H&#x0026;E-stained colon biopsy images contain pink-colored connecting tissues, purple-colored nuclei, and white-colored epithelial cells and lumen [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-10">10</xref>]. The structure of a normal/healthy colon biopsy image has a definite glandular structure for the white-colored epithelial cells [<xref ref-type="bibr" rid="ref-9">9</xref>,<xref ref-type="bibr" rid="ref-36">36</xref>], as shown in <xref ref-type="fig" rid="fig-1">Fig. 1a</xref>. However, this definite structure is distorted when cancer occurs as the white-colored epithelial cells and lumen gradually combine with the pink-colored connecting tissues, and the deformation increases as the grade of cancer advances. The differentiability of malignant cells is quantified by three colon cancer grades wherein their color composition and texture vary [<xref ref-type="bibr" rid="ref-36">36</xref>]. The glandular shape is almost maintained in well-differentiated tumors (<xref ref-type="fig" rid="fig-1">Fig. 1b</xref>), whereas the moderately differentiable grade differs from the normal shape (<xref ref-type="fig" rid="fig-1">Fig. 1c</xref>). The epithelial cells that form the glandular border irregularly scatter in poorly differentiated tumors, making it difficult to determine individual glands border (<xref ref-type="fig" rid="fig-1">Fig. 1d</xref>). Thus, developing a framework that classifies H&#x0026;E-stained images into four grades: normal, well, moderate, and poor, is difficult.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Four classes of colon biopsy images: (a) normal, (b) well, (c) moderate, and (d) poor</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-1.png"/>
</fig>
<p>The proposed framework is evaluated using the colon pathological image data obtained from four independent sources (two collected from Indian hospitals and two benchmark datasets) from different locations and at different microscope magnifications at which the pathologist observed the tissue sample:</p>
<p><inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mo>&#x2219;</mml:mo></mml:math></inline-formula> <italic>Ishita Pathology Center dataset</italic>: 1200 images at a resolution of 640 <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 480 were collected from H&#x0026;E-stained colon biopsy samples of 5&#x2013;6 <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mi>&#x03BC;</mml:mi></mml:math></inline-formula>mm thick tissue section slides from IPC, Allahabad, India for magnifications of 4X, 10X, and 40X. For each grade under a particular magnification, there are 100 images (normal = 100, well = 100, moderate = 100, and poor = 100). A Magcam CD5 with Olympus CX33 was used to capture the images. Dr. Ranjana Srivastava, the Senior Consultant at IPC, analyzed the H&#x0026;E slides and prepared the ground truth labels for the dataset.</p>
<p><inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mo>&#x2219;</mml:mo></mml:math></inline-formula> <italic>AMC dataset</italic>: 840 images at a resolution of 640 <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 480 were collected from H&#x0026;E-stained colon biopsy samples of 5&#x2013;6 <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mi>&#x03BC;</mml:mi></mml:math></inline-formula>mm thick tissue section slides from the Department of Pathology, Aster Medcity (AMC), Kochi, India, for magnifications of 10X, 20X, and 40X. For each grade under a particular magnification, there are 70 images (normal = 70, well = 70, moderate = 70, and poor = 70). A NIS element viewer microscope was used to view the slides, and a Nikon eclipse Ci was used to capture the images. Dr. Sarah Kuruvila (Former Senior Consultant, Pathology Department, Aster Medcity, Kochi, India) and Dr. Shahin Hameed (Consultant Pathologist, MVR Cancer Center and Research Institute, Poolacode, Kerala, India) analyzed the H&#x0026;E slides of the colon biopsy. They prepared the dataset and provided the ground truth labels.</p>
<p><inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mo>&#x2219;</mml:mo></mml:math></inline-formula> <italic>GlaS dataset</italic> [<xref ref-type="bibr" rid="ref-18">18</xref>]: 165 images acquired at a 20X magnification with 640 <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 480 resolution were collected from the GlaS dataset. Images were labeled by an expert pathologist as normal = 74, moderate = 47, moderate-to-poor = 20, and poor = 24.</p>
<p><inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mo>&#x2219;</mml:mo></mml:math></inline-formula> <italic>IMEDIATREAT dataset</italic> [<xref ref-type="bibr" rid="ref-31">31</xref>]: 357 10X-magnified images were acquired at a resolution of 800 <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 600 with 62 normal (G0) records, 96 of the first grade (G1), 99 of the second grade (G2), and 100 of the third grade (G3).</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Images of normal samples from the IPC and AMC datasets at different magnifications</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-2.png"/>
</fig>
<p>The pathologist followed the eighth edition of the manual for tumor node metastasis (TNM) defined by the American Joint Committee on Cancer (AJCC) for the preparation and ground truth labeling of IPC and AMC datasets [<xref ref-type="bibr" rid="ref-37">37</xref>]. The images of GlaS and IMEDIATREAT datasets were labeled as normal, well, moderate, and poor, respectively, and resized to 640 <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 480 resolution to maintain the uniformity of the images and labels across the four datasets. <xref ref-type="fig" rid="fig-2">Fig. 2</xref> shows normal colon biopsy images acquired from the IPC and AMC datasets at various microscopic magnifications, providing an understanding of how the colon biopsy images vary across different magnification and staining conditions.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Proposed Methodology</title>
<p>The schematic framework of the proposed colon cancer grading framework comprises three modules: (i) preprocessing, (ii) feature extraction, and (iii) classification, as shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, which is discussed in detail in the following subsections.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Block diagram of the proposed framework</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-3.png"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Pre-Processing Module</title>
<p>In the first phase of preprocessing, stain normalization [<xref ref-type="bibr" rid="ref-5">5</xref>] and contrast enhancement [<xref ref-type="bibr" rid="ref-38">38</xref>] are conducted to increase image quality. As the input images are from different datasets and slides that undergo distinct staining and illumination conditions, stain normalization is performed, wherein there is a reference image (chosen by the expert pathologist) to which all other images need to be stain-normalized. <xref ref-type="fig" rid="fig-4">Fig. 4a</xref> shows the input image that has to be stain-normalized with respect to the reference image (<xref ref-type="fig" rid="fig-4">Fig. 4b</xref>) and stain-normalized image (<xref ref-type="fig" rid="fig-4">Fig. 4c</xref>). Thus, post stain normalization, all input colon biopsy images are further contrast-enhanced. Later, for extracting texture features, stain-normalized contrast-enhanced images are converted to grayscale.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Stain Normalization: (a) raw image; (b) reference image; and (c) normalized image</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-4.png"/>
</fig>
<p>The components of colon biopsy images are typically distinguished as nuclei in purple color, connecting tissues in pink, and the epithelial and lumen in white color [<xref ref-type="bibr" rid="ref-9">9</xref>&#x2013;<xref ref-type="bibr" rid="ref-11">11</xref>]. Therefore, to obtain these clusters, K-means clustering [<xref ref-type="bibr" rid="ref-39">39</xref>] was performed on stain-normalized contrast-enhanced images with K = 3. The white cluster obtained from K-means is considered for morphological feature extraction as the lumen and epithelial cells constitute the geometric parts and undergo distortion as the cancer grade progresses [<xref ref-type="bibr" rid="ref-10">10</xref>,<xref ref-type="bibr" rid="ref-11">11</xref>]. <xref ref-type="fig" rid="fig-5">Fig. 5a</xref> shows the preprocessed image that undergoes K-means clustering and results in the pink (<xref ref-type="fig" rid="fig-5">Fig. 5b</xref>), purple (<xref ref-type="fig" rid="fig-5">Fig. 5c</xref>), and white clusters (<xref ref-type="fig" rid="fig-5">Fig. 5d</xref>).</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>K-means clustering, K = 3: (a) preprocessed image; (b) pink cluster; (c) purple cluster; and (d) white cluster</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-5.png"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Feature Extraction Module</title>
<p>The variation in texture and color across various magnified images and grades of cancer must be captured using a proper feature set. In the feature extraction phase, the various features extracted from the image were combined to form a novel, rich hybrid feature set to categorize the colon images into four classes. Three significant extracted features are texture, color, and morphology. The texture feature vector, including cartoon texture features, Gabor wavelet, and wavelet moments, is extracted from the grayscale preprocessed image, whereas color features such as HSV histogram, color auto-correlogram, and color moments are extracted from the preprocessed stain-normalized contrast-enhanced image. The morphological features are extracted from the white cluster obtained post-K-means clustering. These feature vectors are then unified to form a rich hybrid feature set grading colon biopsy images at various magnifications.</p>
<sec id="s3_2_1">
<label>3.2.1</label>
<title>Texture Feature</title>
<p>The preprocessed grayscale image was used to extract the following texture features.</p>
<list list-type="bullet">
<list-item><p><italic>Cartoon Texture Feature</italic>: These features primarily contain geometric parts, such as piecewise-smooth regions and edge contours on a large scale. They utilize both local and nonlocal systems, which can exploit similar patches for textures&#x2019; sparse representation. More rich features are required when malignancy changes according to grade and microscopic magnifications, in which case cartoon features provide better edge detection quality. Further, cartoon features extract more detailed texture by considering the difference between the original image and its cartoon component. As the different grades differ in the structures, the structural deformities could be measured irrespective of the magnification with cartoon texture features as the images are decomposed in the temporal domain. Thus, the cartoon image <italic>c</italic>(<italic>x</italic>) and texture image <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:mi>t</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> are obtained from <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref> for an image <italic>I</italic> for every pixel <italic>x</italic> [<xref ref-type="bibr" rid="ref-40">40</xref>].</p></list-item>
</list>
<disp-formula id="eqn-1">
<label>(1)</label>

<mml:math id="mml-eqn-1" display="block"><mml:mi>c</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x03C9;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03BB;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow></mml:msub><mml:mo>*</mml:mo><mml:mi>I</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>&#x03C9;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03BB;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mspace width=".3em" /><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>d</mml:mi><mml:mspace width=".3em" /><mml:mi>t</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>I</mml:mi><mml:mo>-</mml:mo><mml:mi>c</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>
<p>here the weight function <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mi>&#x03C9;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow class="cases"><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd class="array" columnalign="left"><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mspace width="1em" class="quad"/></mml:mtd><mml:mtd class="array" columnalign="left"><mml:mi>x</mml:mi><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd class="array" columnalign="left"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mspace width="1em" class="quad"/></mml:mtd><mml:mtd class="array" columnalign="left"><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2264;</mml:mo><mml:mi>x</mml:mi><mml:mo>&#x2264;</mml:mo><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd class="array" columnalign="left"><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mspace width="1em" class="quad"/></mml:mtd><mml:mtd class="array" columnalign="left"><mml:mi>x</mml:mi><mml:mo>&#x2265;</mml:mo><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr> </mml:mtable></mml:mrow><mml:mo></mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, <italic>a</italic><sub>1</sub>, <italic>a</italic><sub>2</sub> are constants, and <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:msub><mml:mrow><mml:mi>&#x03BB;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>L</mml:mi><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:mi>L</mml:mi><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow></mml:msub><mml:mo>*</mml:mo><mml:mi>I</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>L</mml:mi><mml:mi>T</mml:mi><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:math></inline-formula>, where local total variation (<italic>LTV</italic>) is obtained through convolution with the gradient norm of the image (<italic>I</italic>) and the low-pass filtered image <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow></mml:msub><mml:mo>*</mml:mo><mml:mi>I</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. Thus, the extracted the cartoon feature vector is of length 480.</p>
<list list-type="bullet">
<list-item><p><italic>Gabor wavelets</italic>: To consider the uncertainty between the time and frequency resolution, the Gabor function provides the lower bound and performs the best analytical resolution in the joint domain [<xref ref-type="bibr" rid="ref-40">40</xref>]. As colon images&#x2019; malignancy degrades cell structure, Gabor features provide more information on edges and corners. For a given image <italic>I(x,y)</italic> having size <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mi>P</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>Q</mml:mi></mml:math></inline-formula>, the discrete Gabor wavelet transform with scale (<italic>m</italic> = 0, 1, <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mo>&#x2026;</mml:mo></mml:math></inline-formula>, <italic>M</italic> &#x2212;1) and orientation (<italic>n</italic> = 0, 1, <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mo>&#x2026;</mml:mo></mml:math></inline-formula>, <italic>N</italic> &#x2212;1) is expressed by <xref ref-type="disp-formula" rid="eqn-2">Eq. (2)</xref> [<xref ref-type="bibr" rid="ref-41">41</xref>]:</p></list-item>
</list>
<disp-formula id="eqn-2">
<label>(2)</label>

<mml:math id="mml-eqn-2" display="block"><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mo>&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:munder><mml:munder><mml:mrow><mml:mo>&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:munder><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>-</mml:mo><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi><mml:mo>-</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x03C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where, <italic>s</italic> and <italic>t</italic> are the filter mask size variables, and <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:msubsup><mml:mrow><mml:mi>&#x03C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msubsup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> is the complex conjugate of the generating function <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msub><mml:mrow><mml:mi>&#x03C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> (set of continuous wavelets), here <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:msub><mml:mrow><mml:mi>&#x03C8;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msup><mml:mi>&#x03C8;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>&#x0303;</mml:mo></mml:mover><mml:mo>,</mml:mo><mml:mi>&#x1EF9;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> with <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mi>&#x03C8;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo class="qopname"> exp</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>.</mml:mo><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:mi>W</mml:mi><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, <italic>W</italic> denotes the modulation frequency; <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the standard deviation; <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>&#x0303;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo class="qopname">cos</mml:mo><mml:mi>&#x03B8;</mml:mi><mml:mo>+</mml:mo><mml:mi>y</mml:mi><mml:mo class="qopname">sin</mml:mo><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>and <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mi>&#x1EF9;</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>m</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mi>x</mml:mi><mml:mo class="qopname">sin</mml:mo><mml:mi>&#x03B8;</mml:mi><mml:mo>+</mml:mo><mml:mi>y</mml:mi><mml:mo class="qopname">cos</mml:mo><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>a</italic> &#x003E; 1, <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:mi>&#x03B8;</mml:mi><mml:mo>=</mml:mo><mml:mi>n</mml:mi><mml:mi>&#x03C0;</mml:mi><mml:mo>/</mml:mo><mml:mi>N</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mi>a</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>/</mml:mo><mml:msub><mml:mrow><mml:mi>U</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mrow></mml:msup></mml:math></inline-formula> where <italic>U<sub>h</sub></italic> and <italic>U<sub>l</sub></italic> represent the set of Gabor wavelets. The Gabor feature vector extracted is of length 60.</p>
<list list-type="bullet">
<list-item><p><italic>Wavelet Moments</italic>: Wavelets have the substantial advantage of separating the fine details in a malignant image with respect to its grades to find more localized features in colon grades. Very small wavelets can be used to isolate very fine details in the malignancy of colon images, whereas very large wavelets can identify coarse details. In conjunction with applying Gabor filters on an image with a distinctive orientation at a different scale, the array is obtained as in <xref ref-type="disp-formula" rid="eqn-3">Eq. (3)</xref> [<xref ref-type="bibr" rid="ref-42">42</xref>].</p></list-item>
</list>
<disp-formula id="eqn-3">
<label>(3)</label>

<mml:math id="mml-eqn-3" display="block"><mml:mi>E</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mrow><mml:mo>&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:munder><mml:munder><mml:mrow><mml:mo>&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where, <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>M</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn><mml:mo>;</mml:mo></mml:math></inline-formula> denotes the scale of wavelet transform and <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn><mml:mo>;</mml:mo></mml:math></inline-formula> denotes orientation. In this research, regions that have homogenous texture must be analyzed; therefore, the mean (<inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:msub><mml:mrow><mml:mi>&#x03BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) and standard deviation (<inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) are expressed as <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:msub><mml:mrow><mml:mi>&#x03BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>Q</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula> and <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:msqrt></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>Q</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula> respectively, where <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:mi>P</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>Q</mml:mi></mml:math></inline-formula> represents image size. Feature vector <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03BC;</mml:mi></mml:mrow><mml:mrow><mml:mn>00</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>00</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03BC;</mml:mi></mml:mrow><mml:mrow><mml:mn>01</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>01</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2026;</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>.</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03BC;</mml:mi></mml:mrow><mml:mrow><mml:mn>20</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>20</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> is tabulated using <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:msub><mml:mrow><mml:mi>&#x03BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. The extracted wavelet moments are of length 40.</p>
<p>Combining all of the above-described texture features yields a feature vector of length 580.</p>
</sec>
<sec id="s3_2_2">
<label>3.2.2</label>
<title>Color Features</title>
<p>Features that can capture variation in the color of the images of healthy and malignant colon color cells are essential. The following color features are extracted for the proposed framework.</p>
<list list-type="bullet">
<list-item><p><italic>HSV Histogram</italic>: As the color composition varies for different grades of the colon biopsy images, a color model aims to generalize and standardize the representation of colors in these images. Hence, an image pixel value is converted from the RGB representation to HSV using the formula given in <xref ref-type="disp-formula" rid="eqn-4">Eq. (4)</xref>.</p></list-item>
</list>
<disp-formula id="eqn-4">
<label>(4)</label>

<mml:math id="mml-eqn-4" display="block"><mml:mi>H</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo> cos</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mfrac><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mo>-</mml:mo><mml:mi>G</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mo>-</mml:mo><mml:mi>B</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mo>-</mml:mo><mml:mi>G</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mo>-</mml:mo><mml:mi>B</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>G</mml:mi><mml:mo>-</mml:mo><mml:mi>B</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mspace width="1em"/><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>3</mml:mn><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo>min</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mo>,</mml:mo><mml:mi>G</mml:mi><mml:mo>,</mml:mo><mml:mi>B</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mo>+</mml:mo><mml:mi>G</mml:mi><mml:mo>+</mml:mo><mml:mi>B</mml:mi></mml:mrow></mml:mfrac> <mml:mspace width=".3em" /><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>d</mml:mi><mml:mspace width=".3em" /><mml:mi>V</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>R</mml:mi><mml:mo>+</mml:mo><mml:mi>G</mml:mi><mml:mo>+</mml:mo><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:mfrac></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></disp-formula>
<p>Formally, the color histogram is defined as <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo>,</mml:mo><mml:mi>c</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>H</mml:mi><mml:mo>=</mml:mo><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mi>b</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mo>=</mml:mo><mml:mi>c</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, where <italic>H</italic>, <italic>S</italic>, and <italic>V</italic> represent the color bands in the chosen color space (HSV), and <italic>N</italic> represents the number of dots in the image. The dimension of the histogram was reduced via the Kherfi et al. [<xref ref-type="bibr" rid="ref-43">43</xref>] solution. The color space was deconstructed into 27 subspaces by dividing each color strip&#x2019;s intensities into three equal parts. The result is a vector of only 27 cells.</p>
<list list-type="bullet">
<list-item><p><italic>Color Auto-correlogram</italic>: This three-dimensional histogram characterizes the color distribution and spatial correlation between color pairs. The first and second dimensions of the histogram represent the colors of any pair of pixels, and the third dimension represents the spatial distance between them [<xref ref-type="bibr" rid="ref-44">44</xref>]. A color correlogram can be treated as a table indexed by color pairs, where the <italic>k</italic><sup><italic>th</italic></sup> entry for (<italic>i</italic>, <italic>j</italic>) specifies the probability that a color pixel <italic>j</italic> is at a distance <italic>k</italic> from another color pixel <italic>i</italic> in the image. Let <italic>H</italic> be the set of pixels of an image and <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub></mml:math></inline-formula> be the set of pixels of color <italic>c</italic>(<italic>j</italic>); then, the image&#x2019;s correlogram is defined as in <xref ref-type="disp-formula" rid="eqn-5">Eq. (5)</xref>.</p></list-item>
</list>
<disp-formula id="eqn-5">
<label>(5)</label>

<mml:math id="mml-eqn-5" display="block"><mml:msubsup><mml:mrow><mml:mi>&#x03B3;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mi>&#x03F5;</mml:mi><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where, <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>, <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:mi>k</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>d</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:math></inline-formula> is the distance between pixels <italic>p</italic><sub>1</sub> and <italic>p</italic><sub>2</sub> and <italic>p<sub>r</sub></italic> is the probability function. The extracted color auto-correlogram feature vector is of length 64.</p>
<list list-type="bullet">
<list-item><p><italic>Color Moments</italic>: If the value of the <italic>i</italic><sup><italic>th</italic></sup> color channel at the <italic>j</italic><sup><italic>th</italic></sup> image pixel is <italic>I<sub>ij</sub></italic>, and the number of pixels is <italic>N</italic>, then the index entries related to this color channel and the color model <italic>r</italic> are known as the color moments defined as in <xref ref-type="disp-formula" rid="eqn-6">Eq. (6)</xref> [<xref ref-type="bibr" rid="ref-11">11</xref>].</p></list-item>
</list>
<disp-formula id="eqn-6">
<label>(6)</label>

<mml:math id="mml-eqn-6" display="block"><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:mstyle displaystyle='true'><mml:munderover><mml:mrow><mml:mo>&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo lspace='0pt' rspace='0pt'>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle></mml:mstyle><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mspace width=".3em" /><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>d</mml:mi><mml:mspace width=".3em" /><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msqrt><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:mstyle displaystyle='true'><mml:munderover><mml:mrow><mml:mo>&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo lspace='0pt' rspace='0pt'>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover></mml:mstyle></mml:mstyle><mml:msup><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:msqrt></mml:math></disp-formula>
<p>here <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2264;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x2264;</mml:mo><mml:mn>3</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> presents the average color (mean) of the region <italic>r</italic>; <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the standard deviation of the color model r and the extracted color features are given by the feature vector <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2026;</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>.</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math></inline-formula>. Color moments are thus extracted for the RGB and HSV color model and the feature vector is of length 12.</p>
<p>The three-color features, when concatenated, yield a feature-length of 103.</p>
</sec>
<sec id="s3_2_3">
<label>3.2.3</label>
<title>Morphological Features</title>
<p>These features are extracted to quantify the shape of the white cluster components because grading affects this cluster, wherein the distortions become severe as the grade progresses. These features are extracted from the white cluster&#x2019;s binary form obtained after K-means clustering [<xref ref-type="bibr" rid="ref-10">10</xref>]. Morphological operations, erosion, and dilation were performed on the cluster, and connected components were identified. Based on these connected components, morphological descriptors such as area, perimeter, eccentricity, Euler number, extent, orientation, compactness, and major and minor axis lengths are tabulated. The average morphological values were then determined using all connected cluster components [<xref ref-type="bibr" rid="ref-9">9</xref>], where the morphological features were of length 9.</p>
<p>A rich hybrid feature set is generated by concatenating all individual features with 692 as the feature-length from all the extracted texture, color, and morphological features.</p>
</sec>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Classification Module</title>
<p>The generated hybrid feature set was formulated via 10-fold cross-validation [<xref ref-type="bibr" rid="ref-45">45</xref>] and classified into four classes with ensemble RF optimized using the Bayesian optimization algorithm (BOA); majority voting was implemented to predict the samples. RF classifier is commonly used in medical applications due to its high predictive precision, management of input data at various scales, and its ability to decrease overfitting features [<xref ref-type="bibr" rid="ref-46">46</xref>&#x2013;<xref ref-type="bibr" rid="ref-48">48</xref>]. Hyperparameter tuning with Bayesian reasoning aid will minimize the time taken to achieve the optimal parameters and yield better results in test set generalization [<xref ref-type="bibr" rid="ref-49">49</xref>].</p>
<p>
<statement>
<label>Algorithm 1: </label>
<title>Optimization algorithm of the Bayesian method <inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x00D8;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mn>1</mml:mn></mml:mstyle><mml:mo>:</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></title>
<p>Input: Target function <italic>f</italic><sup><italic>F</italic></sup>; Limit <italic>N</italic>; Hyperparameter space <inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:mi>&#x00D8;</mml:mi></mml:math></inline-formula>; initial design <inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>:</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>&#x2329;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x232A;</mml:mo></mml:mrow></mml:math></inline-formula> Output: Best hyperparameter obtained <inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:msup><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup></mml:math></inline-formula></p>
<p><list list-type="order">
<list-item><p>For <inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:mi>i</mml:mi><mml:mo>&#x2190;</mml:mo><mml:mn>1</mml:mn><mml:mspace width=".3em" /><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mspace width=".3em" /><mml:mi>n</mml:mi></mml:math></inline-formula> do <inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2190;</mml:mo></mml:math></inline-formula> evaluate <inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p></list-item>
<list-item><p>For <inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:mi>j</mml:mi><mml:mo>&#x2190;</mml:mo><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mspace width=".3em" /><mml:mi>t</mml:mi><mml:mi>o</mml:mi><mml:mi>N</mml:mi></mml:math></inline-formula> do steps 3,4,5</p></list-item>
<list-item><p><inline-formula id="ieqn-54"><mml:math id="mml-ieqn-54"><mml:mi>M</mml:mi><mml:mo>&#x2190;</mml:mo></mml:math></inline-formula> fit model on performance data <inline-formula id="ieqn-55"><mml:math id="mml-ieqn-55"><mml:msub><mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo lspace='0pt' rspace='0pt'>&#x2329;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x232A;</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></p></list-item>
<list-item><p>Select <inline-formula id="ieqn-56"><mml:math id="mml-ieqn-56"><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mo class="qopname"> arg</mml:mo><mml:mo class="qopname">&#x03B8;&#x2208;&#x00D8;</mml:mo><mml:mo class="qopname">max</mml:mo><mml:mspace width=".3em" /><mml:mi>a</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p></list-item>
<list-item><p><inline-formula id="ieqn-57"><mml:math id="mml-ieqn-57"><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2190;</mml:mo></mml:math></inline-formula> evaluate <inline-formula id="ieqn-58"><mml:math id="mml-ieqn-58"><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula></p></list-item>
<list-item><p>Return <inline-formula id="ieqn-59"><mml:math id="mml-ieqn-59"><mml:msup><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:mo>&#x2208;</mml:mo><mml:mo class="qopname"> arg</mml:mo><mml:mo class="qopname">min</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo class="qopname">&#x2026;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula></p></list-item>
</list></p>
</statement></p>
<p>Hyperparameter tuning of an RF of decision trees is achieved using quantile error (QE), a parameter tuned for minimizing the classification error. It is required for multidimensional data such as histopathological images and Bayesian optimization [<xref ref-type="bibr" rid="ref-50">50</xref>,<xref ref-type="bibr" rid="ref-51">51</xref>]. <inline-formula id="ieqn-60"><mml:math id="mml-ieqn-60"><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x2026;</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are the hyperparameters of the decision tree, <inline-formula id="ieqn-61"><mml:math id="mml-ieqn-61"><mml:msub><mml:mrow><mml:mi>&#x00D8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mstyle mathvariant="normal"><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:mi>d</mml:mi></mml:mstyle><mml:mspace width=".3em" /><mml:msub><mml:mrow><mml:mi>&#x00D8;</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, denotes the respective domains, and <italic>n</italic> represents the number of hyperparameters. The algorithm hyperparameter space is defined as <inline-formula id="ieqn-62"><mml:math id="mml-ieqn-62"><mml:mi>&#x00D8;</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x00D8;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x00D8;</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:math></inline-formula> When trained with <inline-formula id="ieqn-63"><mml:math id="mml-ieqn-63"><mml:mi>&#x03B8;</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>&#x00D8;</mml:mi></mml:math></inline-formula> on data <inline-formula id="ieqn-64"><mml:math id="mml-ieqn-64"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="italic"><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mstyle></mml:mrow></mml:msub></mml:math></inline-formula>, the QE on data <inline-formula id="ieqn-65"><mml:math id="mml-ieqn-65"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="italic"><mml:mi>v</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi></mml:mstyle></mml:mrow></mml:msub></mml:math></inline-formula> is <inline-formula id="ieqn-66"><mml:math id="mml-ieqn-66"><mml:mi>Q</mml:mi><mml:mi>E</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="italic"><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mstyle></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="italic"><mml:mi>v</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi></mml:mstyle></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:math></inline-formula> Using <italic>k</italic>-fold cross-validation, the hyperparameter optimization for the given dataset <italic>F</italic> is formulated to minimize as in <xref ref-type="disp-formula" rid="eqn-7">Eq. (7)</xref>:</p>
<p><disp-formula id="eqn-7">
<label>(7)</label>
<mml:math id="mml-eqn-7" display="block"><mml:msup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo> min</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mrow><mml:mo>&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo lspace='0pt' rspace='0pt'>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mi>Q</mml:mi><mml:mi>E</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="italic"><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="italic"><mml:mi>v</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>d</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula></p>
<fig id="fig-13">
<graphic mimetype="image" mime-subtype="png" xlink:href="fig-13.png"/>
</fig>
<p>As described in Algorithm 1, Bayesian optimization begins with function <italic>f</italic> at <italic>N</italic> values in the initial design and recording (input, output) pairs <inline-formula id="ieqn-67"><mml:math id="mml-ieqn-67"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mo lspace='0pt' rspace='0pt'>&#x2329;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>&#x232A;</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>. Then, it iterates the operation in three phases: (1) fit a probabilistic model <italic>M</italic> to the considered (input, output) pairs; (2) use the probabilistic model <italic>M</italic> to select a promising input <inline-formula id="ieqn-68"><mml:math id="mml-ieqn-68"><mml:mi>&#x03B8;</mml:mi></mml:math></inline-formula> to evaluate the next by quantifying the desirability of obtaining the function value at arbitrary inputs <inline-formula id="ieqn-69"><mml:math id="mml-ieqn-69"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>&#x00D8;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> through an acquisition function <inline-formula id="ieqn-70"><mml:math id="mml-ieqn-70"><mml:mi>a</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>; and (3) evaluate the function at the new input <inline-formula id="ieqn-71"><mml:math id="mml-ieqn-71"><mml:mi>&#x03B8;</mml:mi></mml:math></inline-formula>.</p>
<p>The role of the acquisition function <inline-formula id="ieqn-72"><mml:math id="mml-ieqn-72"><mml:mi>a</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> is trade-off exploration in hyperparameter regions where the model <italic>M</italic> is uncertain with exploitation in regions with low predicted QE. The acquisition function&#x2019;s expected improvement over the best input found thus far [<xref ref-type="bibr" rid="ref-46">46</xref>] is represented by <xref ref-type="disp-formula" rid="eqn-8">Eq. (8)</xref>.</p>
<p><disp-formula id="eqn-8">
<label>(8)</label>

<mml:math id="mml-eqn-8" display='block'><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>E</mml:mi><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:mrow><mml:msubsup><mml:mo>&#x222B;</mml:mo><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>&#x221E;</mml:mi></mml:mrow><mml:mi>&#x221E;</mml:mi></mml:msubsup><mml:mrow><mml:mi>max</mml:mi></mml:mrow></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>y</mml:mi><mml:mo>*</mml:mo></mml:msup><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mi>p</mml:mi><mml:mi>M</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>y</mml:mi><mml:mo>&#x007C;</mml:mo><mml:mi>&#x03B8;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:msub><mml:mi>d</mml:mi><mml:mi>y</mml:mi></mml:msub></mml:mrow></mml:math></disp-formula></p>
<p><xref ref-type="fig" rid="fig-6">Fig. 6</xref> visualizes the change in the objective function value versus the number of function evaluations for the Bayesian optimized RF. Therein, the objective function reaches its global minimum within 30 iterations at maximum. It reiterates the BOA&#x2019;s efficiency in optimizing the considered algorithms.</p>
<p>RF parameters were optimized using the BOA. The training set was constructed using hybrid feature variables obtained using the proposed method. Before the RF model was trained, the RF parameters were determined, including the number of trees, <inline-formula id="ieqn-73"><mml:math id="mml-ieqn-73"><mml:mstyle mathvariant="italic"><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>e</mml:mi></mml:mstyle></mml:math></inline-formula>; the number of leaves per tree, <italic>nleaf</italic>; and the number of random variables used for each node split, <italic>mtry</italic>. If minimum classification loss is considered the goal, the number of decision trees can drastically increase. The two parameters were optimized to improve classification accuracy. If the <italic>nleaf</italic> value is too large, it results in overfitting; if the <italic>nleaf</italic> value is too small, it results in underfitting. The RF parameters <italic>nleaf</italic> and <italic>mtry</italic> were tuned using BOA and set with <italic>ntree</italic> = 300, <inline-formula id="ieqn-74"><mml:math id="mml-ieqn-74"><mml:mi>n</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>f</mml:mi><mml:mi>&#x03F5;</mml:mi></mml:math></inline-formula> [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-20">20</xref>], and <inline-formula id="ieqn-75"><mml:math id="mml-ieqn-75"><mml:mi>m</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>y</mml:mi><mml:mi>&#x03F5;</mml:mi></mml:math></inline-formula> [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-10">10</xref>]. The objective function of BOA is the QE. <xref ref-type="fig" rid="fig-6">Fig. 6</xref> shows the objective function model and shows the relationship between function evaluations and the minimum objective. The optimized RF parameters were calculated as <italic>nleaf</italic> = 7 and <italic>mtry</italic> = 5, and the observed minimum of the objective function was 0.005.</p>
<p>Once the RF classifiers were optimized, determining the number of binary Bayesian optimized RF classifiers was important for appropriate four-class classification, as shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>. Hence, there is a need to build the <inline-formula id="ieqn-76"><mml:math id="mml-ieqn-76"><mml:mfrac><mml:mrow><mml:mi>N</mml:mi><mml:mo>*</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac></mml:math></inline-formula> Bayesian optimized RF classifiers: one classifier to distinguish each pair of classes <italic>i</italic> and <italic>j</italic>, where <italic>N</italic> is the number of classes. Let <italic>f<sub>ij</sub></italic> be the classifier where class <italic>i</italic> represents positive examples and class <italic>j</italic> represents negative examples, where <italic>f<sub>ji</sub></italic> = &#x2212; <italic>f<sub>ij</sub></italic> classify using <inline-formula id="ieqn-77"><mml:math id="mml-ieqn-77"><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo class="qopname"> arg</mml:mo><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo class="qopname">&#x2211;</mml:mo> </mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. If the binary classification models predict a numerical class membership, such as a probability, then the <inline-formula id="ieqn-78"><mml:math id="mml-ieqn-78"><mml:mstyle mathvariant="italic"><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mstyle></mml:math></inline-formula> of the sum of the scores, which is the class with the largest sum score, is predicted as the class label.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Bayesian optimized random forest (a) objective function model and (b) minimum objective <italic>vs.</italic> number of function evaluations</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-6.png"/>
</fig>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Results</title>
<p>In this section, first, the performance measures used to evaluate the proposed framework are discussed. Later, the results of the proposed framework are analyzed at five levels.</p>
<sec id="s4_1">
<label>4.1</label>
<title>Performance Measures</title>
<p>The proposed system is quantitatively evaluated based on performance metrics such as accuracy, error rate, sensitivity, specificity, precision, false-positive rate, F-score, Mathew correlation coefficient (MCC), and kappa statistics described in <xref ref-type="table" rid="table-1">Tab. 1</xref>. Accuracy and error rate is measured in percentage, MCC varies from &#x2212;1 to +1, and rest all measures scale from 0&#x2013;1 (1 is best and 0 worst) [<xref ref-type="bibr" rid="ref-52">52</xref>]. A <inline-formula id="ieqn-79"><mml:math id="mml-ieqn-79"><mml:mn>4</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>4</mml:mn></mml:math></inline-formula> confusion matrix with true positive (TP), false positive (FP), true negative (TN), and false negative (FN) is used to tabulate the performance measures.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Performance evaluation measures</title>
</caption>

<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Performance measures</th>
<th>Formula</th>
<th>Description</th>
<th></th>
</tr>
</thead>
<tbody>
<tr>
<td>Accuracy</td>
<td><inline-formula id="ieqn-80"><mml:math id="mml-ieqn-80"><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x00D7;</mml:mo><mml:mn>100</mml:mn></mml:math></inline-formula></td>
<td>The classifiers capability to classify the samples</td>
<td/>
</tr>
<tr>
<td>Error rate</td>
<td/>
<td>100-Accuracy</td>
<td/>
</tr>
<tr>
<td>Sensitivity</td>
<td><inline-formula id="ieqn-81"><mml:math id="mml-ieqn-81"><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula></td>
<td>The classifier&#x2019;s capability to identify the positive samples.</td>
<td/>
</tr>
<tr>
<td>Specificity</td>
<td><inline-formula id="ieqn-82"><mml:math id="mml-ieqn-82"><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula></td>
<td>The classifier&#x2019;s capability to identify the positive samples.</td>
<td/>
</tr>
<tr>
<td>Precision</td>
<td><inline-formula id="ieqn-83"><mml:math id="mml-ieqn-83"><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula></td>
<td>The actual positives among the positive predicted samples.</td>
<td/>
</tr>
<tr>
<td>False positive rate</td>
<td/>
<td>1-Specificity</td>
<td/>
</tr>
<tr>
<td>F-score</td>
<td><inline-formula id="ieqn-84"><mml:math id="mml-ieqn-84"><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mfrac><mml:mrow><mml:mstyle mathvariant="italic"><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mstyle><mml:mo>&#x00D7;</mml:mo><mml:mstyle mathvariant="italic"><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle mathvariant="italic"><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mstyle><mml:mo>+</mml:mo><mml:mstyle mathvariant="italic"><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mstyle></mml:mrow></mml:mfrac></mml:math></inline-formula></td>
<td>Recall and Precision&#x2019;s weighted average.</td>
<td/>
</tr>
<tr>
<td>MCC</td>
<td><inline-formula id="ieqn-85"><mml:math id="mml-ieqn-85"><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>-</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:math></inline-formula></td>
<td>Observed and predicted classifications&#x2019; correlation coefficient</td>
<td/>
</tr>
<tr>
<td>Kappa statistic</td>
<td><inline-formula id="ieqn-86"><mml:math id="mml-ieqn-86"><mml:mfrac><mml:mrow><mml:mstyle mathvariant="italic"><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi></mml:mstyle><mml:mspace width=".3em" /><mml:mstyle mathvariant="italic"><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mstyle><mml:mo>-</mml:mo><mml:mstyle mathvariant="italic"><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi></mml:mstyle><mml:mspace width=".3em" /><mml:mstyle mathvariant="italic"><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mstyle mathvariant="italic"><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi></mml:mstyle><mml:mspace width=".3em" /><mml:mstyle mathvariant="italic"><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mstyle></mml:mrow></mml:mfrac></mml:math></inline-formula></td>
<td>It shows how the instances categorized by the classifier corresponds to the records that were labeled as ground truth.</td>
<td/>
</tr>
</tbody>
</table>
</table-wrap>
<p>The overall MCC is determined using the technique macro-averaging for a multi-class classification. Assume 1, 2, 3, and 4 are four categories that classify the samples. Then, with the <inline-formula id="ieqn-87"><mml:math id="mml-ieqn-87"><mml:mn>4</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>4</mml:mn></mml:math></inline-formula> confusion matrix, TP, TN, FP, and FN are computed as: TP = TP1 + TP2 + TP3 + TP4; TN = TN1 + TN2 + TN3 + TN4; FP = FP1 + FP2 + FP3 + FP4; and FN = FN1 + FN2 + FN3 + FN4. The cumulative MCC is estimated using these values.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Experimental Results and Analysis</title>
<p>This section analyzes the efficiency of the proposed method through different datasets and examines the findings. The results of the proposed framework were analyzed in five phases. (1) The first phase of analysis included the evaluation of the magnification-independent framework across various datasets; (2) In the second step, the model was generalized for which evaluation was done using one dataset training and another dataset testing; (3) The third phase comprised the performance analysis of the proposed framework under each considered magnification; (4) In the fourth phase, the performance and interpretation of features were analyzed; and (5) In the fifth phase of analysis, the proposed framework was compared with existing techniques, on the benchmark datasets.</p>
<sec id="s4_2_1">
<label>4.2.1</label>
<title>Performance of the Proposed Colon Cancer Grading Framework</title>
<p>The proposed four-class colon cancer grading framework was evaluated using four different datasets, including various magnifications. To evaluate the proposed framework&#x2019;s magnification-independent nature, for training and testing, colon biopsy images of various microscopic magnifications were considered from IPC (4X, 10X, and 40X microscope magnifications) and AMC (10X, 20X, and 40X microscope magnifications) datasets. <xref ref-type="table" rid="table-2">Tab. 2</xref> summarizes the performance measures of the proposed model for different datasets.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Performance evaluation measures of the proposed framework on different datasets</title>
</caption>

<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Performance measures</th>
<th>IPC</th>
<th>AMC</th>
<th>GlaS</th>
<th>IMEDIATREAT</th>
</tr>
</thead>
<tbody>
<tr>
<td>Accuracy</td>
<td>97.250</td>
<td>94.400</td>
<td>97.580</td>
<td>99.160</td>
</tr>
<tr>
<td>Error rate</td>
<td>2.7500</td>
<td>5.6000</td>
<td>2.4200</td>
<td>0.0840</td>
</tr>
<tr>
<td>Sensitivity</td>
<td>0.9725</td>
<td>0.9440</td>
<td>0.9807</td>
<td>0.9923</td>
</tr>
<tr>
<td>Specificity</td>
<td>0.9908</td>
<td>0.9813</td>
<td>0.9907</td>
<td>0.9971</td>
</tr>
<tr>
<td>Precision</td>
<td>0.9731</td>
<td>0.9447</td>
<td>0.9759</td>
<td>0.9923</td>
</tr>
<tr>
<td>False positive rate</td>
<td>0.9902</td>
<td>0.0187</td>
<td>0.0093</td>
<td>0.0029</td>
</tr>
<tr>
<td>F-score</td>
<td>0.9725</td>
<td>0.9441</td>
<td>0.9780</td>
<td>0.9923</td>
</tr>
<tr>
<td>MCC</td>
<td>0.9636</td>
<td>0.9257</td>
<td>0.9690</td>
<td>0.9894</td>
</tr>
<tr>
<td>Kappa statistic</td>
<td>0.9267</td>
<td>0.8508</td>
<td>0.9354</td>
<td>0.9776</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The four-class grading performed with the Bayesian optimized random forest classifier was most accurate for the IMEDIATREAT dataset, with 99.16% accuracy. In contrast, the GlaS, IPC, and AMC datasets were 97.58%, 97.25%, and 94.40% accurate, respectively. The calculated MCC was highest for the IMEDIATREAT dataset, at 0.9894, and the F-score was also higher in the IMEDIATREAT dataset, at 0.9923. The AMC dataset had the lowest MCC value (0.9257). The IMEDIATREAT dataset was most accurate with the proposed system, and the average accuracy calculated for all datasets was 97.09%. Sensitivity is an essential measure in the medical field; hence, the proposed model yields better sensitivity values of 0.9725, 0.9440, 0.9807, and 0.9923 for IPC, AMC, GlaS, and IMEDIATREAT datasets, respectively. Thus, irrespective of various magnified images considered for training and testing with IPC and AMC datasets, the proposed framework is robust across magnifications and datasets.</p>
<p>The <inline-formula id="ieqn-88"><mml:math id="mml-ieqn-88"><mml:mn>4</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>4</mml:mn></mml:math></inline-formula> confusion matrix obtained from the BO-RF ensemble classifier appears in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>, where the rich hybrid feature set is used for the four-class classification. The confusion matrix of the IPC dataset (<xref ref-type="fig" rid="fig-7">Fig. 7a</xref>) demonstrates that TP for the normal class is 98.3%, and the class&#x2019;s misclassifications have occurred with the well class. When considering the well class, 95.7% constitute the TP, and the misclassifications happen with the normal and moderate classes. Similarly, for the moderate class, the misclassifications occur with the well and poor class with a TP of 97.3%. As the poor class structure is entirely different, its misclassification occurs with the moderate and has a TP of 98%. The class-wise analysis of TP for various grades: well (95.7%-IPC, 91%-AMC, 95.7%-Imediatreat), moderate (97.3%-IPC, 94.3%-AMC, 95%-GlaS, 99%-Imediatreat), and poor (96%-IPC, 95.7%-AMC, 100%-GlaS, 100%-Imediatreat) across datasets shows the robustness of the proposed grading irrespective of datasets and magnifications. Further, misclassification occurs with normal and well, well and moderate, and moderate and poor classes as their structure varies little between classes. The minimum misclassifications occur in the poor class as its structure is entirely different from the other classes. The number of FP and FN are minimum for all datasets, thereby boosting the sensitivity. The proposed model uses majority voting with six BO&#x2013;RF, thereby reducing the misclassifications with an average accuracy of &#x00BF;96%.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Confusion matrix plot for the proposed model on different datasets. (a) IPC (b) AMC (c) GlaS (d) IMEDIATREAT 
 
</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-7.png"/>
</fig>
<p>A receiver operating characteristic (ROC) analysis has been conducted; the corresponding results are presented in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>. Each of the classes-normal, well, moderate, and poor, in every dataset demonstrate good ROC as the curve is toward the top left corner even though the respective class rankings vary. The IMEDIATREAT dataset exhibits better ROC for each class as all ROC curves are toward the top left corner. The ROCs across datasets reveal the robustness of the model across multiple magnifications and datasets.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Receiver operating characteristic plot of the proposed framework for four classes on different datasets. (a) IPC (b) AMC (c) GlaS (d) IMEDIATREAT 
 
</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-8.png"/>
</fig>
</sec>
<sec id="s4_2_2">
<label>4.2.2</label>
<title>Performance of the Proposed Model with Training on One Dataset and Testing with Another</title>
<p>The proposed model was trained on one dataset and tested with another dataset and vice versa to assess the proposed model&#x2019;s generalizability. Cross-training and testing ensure the prediction model&#x2019;s performance using an unknown dataset, and the performance measures are illustrated in <xref ref-type="table" rid="table-3">Tab. 3</xref>. The proposed system was evaluated for different training and testing scenarios under all magnifications. The model was trained with the IPC dataset with all magnified images, and it was tested across the AMC, GlaS, and IMEDIATREAT datasets. The highest accuracy (95.80%) was observed on the IMEDIATREAT dataset, and the accuracy on the AMC dataset (91.43%) slightly outperformed that on the GlaS dataset (88.48%). As the training was performed with 4X, 10X, and 40X images, IMEDIATREAT and AMC datasets containing 10X images exhibited considerable outperformance than other datasets, whereas the performance with the GlaS dataset was found to be on the lower side when 20X images were used for testing. Similarly, the AMC dataset comprising 10X, 20X, and 40X images were trained with the system model and tested against the IPC, GlaS, and IMEDIATREAT datasets. When tested, the IPC dataset exhibited the highest accuracy (94.42%) as the testing contained 10X and 40X images, followed by the GlaS (92.73%) and IMEDIATREAT (91.60%) datasets, in that order. GlaS and IMEDIATREAT datasets have shown comparable results as their magnifications were used for training. Subsequently, the GlaS dataset was trained and tested against the IPC, AMC, and IMEDIATREAT datasets. The highest accuracy was 89.88% for the AMC dataset, whereas the IPC dataset yielded a lower accuracy of 86.08%. Compared to other datasets, when trained with the GlaS dataset, the test datasets&#x2019; performance dipped because of the training image sets being few, single magnified, and imbalanced images across the four classes. When the IMEDIATREAT dataset containing 10X images was used for training, and IPC, AMC, and GlaS datasets were used for testing, the highest accuracy was achieved for the GlaS dataset (94.55%) because it contained a single magnification while other datasets contained multiple magnifications for testing.</p>
<p>Analyzing the overall statistical measures for the cross-training and testing outcome from <xref ref-type="table" rid="table-3">Tab. 3</xref> indicates the model&#x2019;s generalization capability across various datasets. When trained with the IPC dataset, the average accuracy was 91.90%. Similarly, when trained with the AMC dataset, the average accuracy was 92.91%, and when trained with GlaS, the average accuracy was 88.16%; IMEDIATREAT yielded an average accuracy of 92.18%.</p>
</sec>
<sec id="s4_2_3">
<label>4.2.3</label>
<title>Performance Analysis of the Proposed Framework under each Magnification</title>
<p>To determine the supremacy of the proposed framework, the analysis under each magnification was performed for IPC and AMC datasets. The model was also tested for cross-training and testing under each magnification across datasets for generalizability.</p>
<p>The proposed magnification-independent framework was evaluated for each magnified image in IPC and AMC datasets. The respective magnified images were considered for training and testing to analyze each magnification&#x2019;s proposed model&#x2019;s performance. <xref ref-type="table" rid="table-4">Tab. 4</xref> illustrates the calculated accuracy, i.e., 94.25%, 96.50%, and 97.50% for the IPC dataset for image magnifications of 4X, 10X, and 40X, respectively. For the IPC dataset, 40X magnification provides higher accuracy than lower magnifications, whereas, in the AMC dataset, a lower magnification of 10X provides higher accuracy (98.57%). F-Scores of 0.9425, 0.9650, and 0.9749 and 0.9857, 0.9643, and 0.9447 are observed on the IPC and AMC datasets for 4X, 10X, and 40X and for 10X, 20X and 40X magnifications, respectively. For the IPC dataset, MCC at 40X was 0.9668, and it was 0.9810 in the AMC dataset at 10X magnification. The difference in data acquisition, lighting, and staining conditions can cause variation in the feature responses, thereby affecting performance across magnifications.</p>
<p><xref ref-type="table" rid="table-5">Tab. 5</xref> demonstrates the proposed model&#x2019;s performance accuracy when trained and tested with independent datasets at different magnifications. The model is trained with one particular magnified image of a dataset and tested with other datasets&#x2019; same magnified images. The cross-training and testing accuracy when the IPC dataset at 10X magnification was trained and tested with IMEDIATREAT was 97.20%, whereas training with IMEDIATREAT and testing with the IPC dataset at 10X magnification yields a lower accuracy (94.50%) than that obtained using the earlier dataset. Similarly, if trained with the AMC dataset at a 20X magnification and tested with GlaS, the system&#x2019;s accuracy was 91.52%, and when the same process was reversed, the accuracy improved to 97.14%. The performance variation is caused by the difference in image acquisition, quality, and staining properties. When comparing the same magnification, such as 40X and training with the AMC dataset, testing with the IPC dataset achieved an accuracy of 97.25%. When training and testing were conducted vice versa, the accuracy of the system was 94.44%. Thus, even with regards to magnifications when the independent datasets are sampled for testing and training, the performance is comparable and demonstrates the model&#x2019;s robustness.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Performance measures of the proposed grading framework with cross-training and testing</title>
</caption>

<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead valign="bottom">
<tr>
<th>Training dataset</th>
<th>Performance measures</th>
<th colspan="4">Testing dataset</th>
</tr>
</thead>
<tbody>
<tr>
<th></th>
<th></th>
<th>IPC</th>
<th>AMC</th>
<th>GlaS</th>
<th>IMEDIATREAT</th>
</tr>
<tr>
<th> IPC</th>
<th>Accuracy</th>
<th>&#x2013;</th>
<th>91.43</th>
<th>88.48</th>
<th>95.80</th>
</tr>
<tr>
<th></th>
<th>Sensitivity</th>
<th>&#x2013;</th>
<th>0.9143</th>
<th>0.8594</th>
<th>0.9527</th>
</tr>
<tr>
<th></th>
<th>Specificity</th>
<th>&#x2013;</th>
<th>0.9714</th>
<th>0.9626</th>
<th>0.9858</th>
</tr>
<tr>
<th></th>
<th>F-score</th>
<th>&#x2013;</th>
<th>0.9140</th>
<th>0.8540</th>
<th>0.9553</th>
</tr>
<tr>
<th></th>
<th>MCC</th>
<th>&#x2013;</th>
<th>0.8855</th>
<th>0.8159</th>
<th>0.9416</th>
</tr>
<tr>
<th>AMC</th>
<th>Accuracy</th>
<th>94.42</th>
<th>&#x2013;</th>
<th>92.73</th>
<th>91.60</th>
</tr>
<tr>
<th></th>
<th>Sensitivity</th>
<th>0.9442</th>
<th>&#x2013;</th>
<th>0.8990</th>
<th>0.9141</th>
</tr>
<tr>
<th></th>
<th>Specificity</th>
<th>0.9814</th>
<th>&#x2013;</th>
<th>0.9766</th>
<th>0.9719</th>
</tr>
<tr>
<th></th>
<th>F-score</th>
<th>0.9442</th>
<th>&#x2013;</th>
<th>0.8959</th>
<th>0.9133</th>
</tr>
<tr>
<th></th>
<th>MCC</th>
<th>0.9256</th>
<th>&#x2013;</th>
<th>0.8724</th>
<th>0.8853</th>
</tr>
<tr>
<th>GlaS</th>
<th>Accuracy</th>
<th>86.08</th>
<th>89.88</th>
<th>&#x2013;</th>
<th>88.52</th>
</tr>
<tr>
<th></th>
<th>Sensitivity</th>
<th>0.8608</th>
<th>0.8988</th>
<th>&#x2013;</th>
<th>0.8798</th>
</tr>
<tr>
<th></th>
<th>Specificity</th>
<th>0.9536</th>
<th>0.9663</th>
<th>&#x2013;</th>
<th>0.9616</th>
</tr>
<tr>
<th></th>
<th>F-score</th>
<th>0.8603</th>
<th>0.8989</th>
<th>&#x2013;</th>
<th>0.8790</th>
</tr>
<tr>
<th></th>
<th>MCC</th>
<th>0.8143</th>
<th>0.8654</th>
<th>&#x2013;</th>
<th>0.8412</th>
</tr>
<tr>
<th>IMEDIATREAT</th>
<th>Accuracy</th>
<th>93.08</th>
<th>88.93</th>
<th>94.55</th>
<th>&#x2013;</th>
</tr>
<tr>
<th></th>
<th>Sensitivity</th>
<th>0.9308</th>
<th>0.8893</th>
<th>0.9345</th>
<th>&#x2013;</th>
</tr>
<tr>
<th></th>
<th>Specificity</th>
<th>0.9769</th>
<th>0.9631</th>
<th>0.9825</th>
<th>&#x2013;</th>
</tr>
<tr>
<th></th>
<th>F-score</th>
<th>0.9307</th>
<th>0.8892</th>
<th>0.9264</th>
<th>&#x2013;</th>
</tr>
<tr>
<th></th>
<th>MCC</th>
<th>0.9077</th>
<th>0.8524</th>
<th>0.9088</th>
<th>&#x2013;</th>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_2_4">
<label>4.2.4</label>
<title>Performance and Interpretation of Features</title>
<p>A quantitative and qualitative evaluation of the proposed framework for individual and combined features with accuracy and F-score is shown in <xref ref-type="fig" rid="fig-9">Figs. 9a</xref> and <xref ref-type="fig" rid="fig-9">9b</xref>, respectively. For the IPC, AMC, GlaS, and IMEDIATREAT datasets, the proposed rich hybrid feature set&#x2019;s average accuracy was 97.25%, 94.44%, 97.58%, and 99.16%, respectively, at the higher side. When individual features were analyzed, the cartoon feature yielded the highest accuracy for the IPC (94.40%) and IMEDIATREAT (97.22%) datasets, and for AMC and GlaS, the highest contributing features varied. Color-moment-based features exhibited a lower accuracy of fit (86.11%) for the IPC and AMC datasets. For the GlaS and IMEDIATREAT datasets, morphological features and wavelets exhibited the lowest system performances of 86.21% and 89.94%, respectively. The texture features contributed more than other features across all datasets for the grading, with accurate data fits of 96.55%, 93.10%, 95.59%, and 96.55% for the IPC, AMC, GlaS, and IMEDIATREAT datasets, respectively. The individual accuracy and F-score for color and morphological features were higher when considered separately rather than when they were combined. An accurate data fit of 90.83% (IPC), 86.11% (AMC), 90.70% (GlaS), and 89.60% (IMEDIATREAT) was found for the combination of color and morphology, which was lower than that obtained when the color and morphology features were considered separately. The texture feature combined with the morphological feature provided the next contributing features with accuracies of 95.59%, 92.86%, 94.85%, and 94.17% across the IPC, AMC, GlaS, and IMEDIATREAT datasets, respectively. Accuracy levels dropped when texture and color were combined. Thus, features, when concatenated, boost accuracy by 1%&#x2013;3%. The accuracy and F-score achieved for the proposed hybrid feature are higher for all datasets when compared with the individual features.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Performance evaluation of the proposed framework across different magnifications for the IPC and AMC datasets</title>
</caption>

<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead valign="bottom">
<tr>
<th>Performance measures</th>
<th colspan="3">IPC</th>
<th colspan="3">AMC</th>
</tr>
</thead>
<tbody>
<tr>
<th></th>
<th>4X</th>
<th>10X</th>
<th>40X</th>
<th>10X</th>
<th>20X</th>
<th>40X</th>
</tr>
<tr>
<th> Accuracy</th>
<th>94.25</th>
<th>96.50</th>
<th>97.50</th>
<th>98.57</th>
<th>96.43</th>
<th>94.64</th>
</tr>
<tr>
<th>Error Rate</th>
<th>5.75</th>
<th>3.50</th>
<th>2.50</th>
<th>1.43</th>
<th>3.57</th>
<th>5.36</th>
</tr>
<tr>
<th>Sensitivity</th>
<th>0.9425</th>
<th>0.9650</th>
<th>0.9750</th>
<th>0.9857</th>
<th>0.9643</th>
<th>0.9464</th>
</tr>
<tr>
<th>Specificity</th>
<th>0.9808</th>
<th>0.9883</th>
<th>0.9917</th>
<th>0.9952</th>
<th>0.9881</th>
<th>0.9821</th>
</tr>
<tr>
<th>Precision</th>
<th>0.9425</th>
<th>0.9682</th>
<th>0.9753</th>
<th>0.9859</th>
<th>0.9643</th>
<th>0.9521</th>
</tr>
<tr>
<th>False Positive Rate</th>
<th>0.0192</th>
<th>0.0117</th>
<th>0.0083</th>
<th>0.0048</th>
<th>0.0119</th>
<th>0.0179</th>
</tr>
<tr>
<th>F-Score</th>
<th>0.9425</th>
<th>0.9650</th>
<th>0.9749</th>
<th>0.9857</th>
<th>0.9643</th>
<th>0.9447</th>
</tr>
<tr>
<th>MCC</th>
<th>0.9233</th>
<th>0.9534</th>
<th>0.9668</th>
<th>0.9810</th>
<th>0.9524</th>
<th>0.9309</th>
</tr>
<tr>
<th>Kappa Statistics</th>
<th>0.8467</th>
<th>0.9067</th>
<th>0.9333</th>
<th>0.9619</th>
<th>0.9048</th>
<th>0.8571</th>
</tr>
</tbody>
</table>
</table-wrap>
 
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Cross-training and testing accuracy for different magnifications</title>
</caption>

<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead valign="bottom">
<tr>
<th>Training in respective magnifications</th>
<th colspan="7">Testing in respective magnifications</th>
</tr>
</thead>
<tbody>
<tr>
<th></th>
<th colspan="3">10X</th>
<th colspan="2">20X</th>
<th colspan="2">40X</th>
</tr>
<tr>
<th></th>
<th>IPC</th>
<th>AMC</th>
<th>IMEDIATREAT</th>
<th>AMC</th>
<th>GlaS</th>
<th>IPC</th>
<th>AMC</th>
</tr>
<tr>
<th> IPC</th>
<th>&#x2013;</th>
<th>95.71</th>
<th>97.20</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
<th>94.44</th>
</tr>
<tr>
<th>AMC</th>
<th>96.75</th>
<th>&#x2013;</th>
<th>93.28</th>
<th>&#x2013;</th>
<th>91.52</th>
<th>97.25</th>
<th>&#x2013;</th>
</tr>
<tr>
<th>GlaS</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
<th>97.14</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
</tr>
<tr>
<th>IMEDIATREAT</th>
<th>94.50</th>
<th>96.43</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
<th>&#x2013;</th>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="fig" rid="fig-10">Fig. 10</xref> illustrates the mosaic plot for the different feature set distributions extracted from different datasets across magnifications. The feature distribution was plotted for the IPC, AMC, and IMEDIATREAT datasets at 10X magnification, the AMC and GlaS datasets at 20X magnification, and the IPC and AMC datasets at 40X magnification. Different grades of colon images yielded variation in the extracted features. In IPC, the healthy colon images showed less variation than other grades. The cartoon features are less sensitive toward magnification variation. They exhibited a symmetrical structure in the mosaic plot for 10X, 20X, and 40X magnifications for different colon cancer image grades. Morphological features changed at different magnifications. An evident difference existed in 10X, 20X, and 40X magnifications for different grades in different datasets. The above mosaic plots indicate feature variation for different grades for colon cancer analysis. Thus, the proposed hybrid features provide a rich classifier platform for better classification of the four-class cancer grading framework across multiple image sources.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>(a) Accuracy and (b) F-Score for individual features and feature combinations on the proposed framework</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-9.png"/>
</fig>
<fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>Feature distribution across different magnifications for different datasets 
 
</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-10.png"/>
</fig>
<fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>Boxplot for Hybrid feature distribution across various datasets 
 
</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-11.png"/>
</fig>
<p>The hybrid feature distribution across datasets in the boxplot from <xref ref-type="fig" rid="fig-11">Fig. 11</xref> shows the system&#x2019;s performance with cross-training and testing. First, the proposed system&#x2019;s hybrid feature is less skewed than other features. Skewness indicates that the data may not be normally distributed. Hence, the extracted hybrid feature has a stable distribution of data for the classifier as a training sample. Second, the IPC and AMC datasets are less skewed in the hybrid-feature-based plot. The median range is in the same range for hybrid features, ranging from 0.056 to 0.070. The IMEDIATREAT dataset variation is more favorable than those in the IPC, AMC, and GlaS datasets for the hybrid features. Thus, the median weights of the notch plots are nearly similar.</p>
<p>Thus, none of the features are individually adequate to separate the four classes; however, multivariate examination through machine learning precisely categorizes normal, well, moderate, and poor classes.</p>
</sec>
<sec id="s4_2_5">
<label>4.2.5</label>
<title>Comparison of the Proposed Model with Existing Techniques</title>
<p>The proposed framework&#x2019;s performance is compared with existing techniques in two aspects, i.e., comparing the activation features extracted from the existing CNN models for four-class classification and comparison with existing techniques on two benchmark datasets.</p>
<p>In the literature [<xref ref-type="bibr" rid="ref-22">22</xref>,<xref ref-type="bibr" rid="ref-23">23</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>], histopathological images were trained over existing CNN models and activation features were extracted for classification as there is scarce annotated medical data, and training from scratch requires extensive data. Commonly used existing CNN models on histopathological image data such as Alexnet [<xref ref-type="bibr" rid="ref-25">25</xref>], VGG-16 [<xref ref-type="bibr" rid="ref-53">53</xref>], Inception v3 [<xref ref-type="bibr" rid="ref-54">54</xref>], and Inception-Resnet v2 [<xref ref-type="bibr" rid="ref-55">55</xref>] are trained on the various colon image datasets to extract high-level features to classify the images into four classes, normal, well, moderate, and poor, with the Bayesian optimized RF classifier, and the comparison with the proposed magnification-independent model is illustrated in <xref ref-type="fig" rid="fig-12">Fig. 12</xref>. The analysis shows that the proposed framework performs better than other CNN models across all datasets regarding the accuracy, sensitivity, specificity, F-score, and MCC. The high-level features extracted from the CNN models are generic features that are not specifically extracted to perform on various image magnifications and grades. The proposed robust hybrid features are meant to extract the varying texture, color, and geometric features across multiple image magnifications and grades. Inception-Resnet v2 is the best CNN model across IPC, AMC, and GlaS datasets, whereas Alexnet performs better on the IMEDIATREAT dataset. System performance across the CNN models differs as the number of levels differs in each of the networks chosen; consequently, system performance varies across the datasets. The proposed magnification-independent multiclass grading framework is a generalized framework that can work across four colon image datasets with multiple magnifications.</p>
<fig id="fig-12">
<label>Figure 12</label>
<caption>
<title>Comparison of the proposed colon cancer grading model with existing CNN models on feature learning on different datasets. (a) IPC (b) AMC (c) GlaS (d) IMEDIATREAT 
 
</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-12.png"/>
</fig>
 
<table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>Comparison of the proposed grading framework with existing techniques on the GlaS and IMEDIATREAT datasets</title>
</caption>

<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Dataset</th>
<th>Paper</th>
<th>Segmentation/ Feature/ Classifier</th>
<th>No. of Images</th>
<th>No. of Classes</th>
<th>Accuracy (%)</th>
<th>Sensitivity</th>
<th>Specificity</th>
<th>F-Score</th>
<th>MCC</th>
</tr>
</thead>
<tbody>
<tr>
<td>GlaS</td>
<td>Awan et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] (2017)</td>
<td>CNN Segmentation/Best alignment matrix/SVM</td>
<td>Healthy = 71, low-grade = 33, high-grade = 35</td>
<td>3</td>
<td>95.33</td>
<td>&#x2013;</td>
<td>0.9716</td>
<td>0.9778</td>
<td>&#x2013;</td>
</tr>
<tr>
<td/>
<td>Saroja et al. [<xref ref-type="bibr" rid="ref-29">29</xref>](2019)</td>
<td>Tree structure generation/Lumen structure/Entropy score computation</td>
<td>Moderate = 47, moderate to poor = 20, poor = 24</td>
<td>3</td>
<td>93.00</td>
<td>0.8076</td>
<td>0.9400</td>
<td>0.9969</td>
<td>0.7900</td>
</tr>
<tr>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
<td/>
</tr>
<tr>
<td/>
<td>Rathore et al. [<xref ref-type="bibr" rid="ref-17">17</xref>] (2019)</td>
<td>Multi-step gland segmentation/Image, gland, patch-based features/Meta-classifier (Linear, RBF, Sigmoid SVM)</td>
<td>Moderate = 47, moderate to poor = 20, poor = 24</td>
<td>3</td>
<td>98.60</td>
<td>0.9730</td>
<td>0.9900</td>
<td>&#x2013;</td>
<td>0.9640</td>
</tr>
<tr>
<td/>
<td><bold>Proposed</bold></td>
<td>High level features (Inception-Resnet v2), Hand-crafted features/Hybrid Features/ensemble BO-RF</td>
<td>Healthy = 74, moderate = 47, moderate to poor = 20, poor = 24</td>
<td>4</td>
<td>97.58</td>
<td>0.9807</td>
<td>0.9907</td>
<td>0.9780</td>
<td>0.9690</td>
</tr>
<tr>
<td>Imediatreat</td>
<td>Boruz et al. [<xref ref-type="bibr" rid="ref-30">30</xref>] (2018)</td>
<td>Intensity-based thresholding/ Morphological features/SVM</td>
<td>Healthy = 62, grade 1 = 96, grade 2 = 99, grade 3 = 100</td>
<td>4</td>
<td>89.75</td>
<td>0.8475</td>
<td>0.9475</td>
<td>0.8412</td>
<td>&#x2013;</td>
</tr>
<tr>
<td/>
<td>Stoean et al. [<xref ref-type="bibr" rid="ref-25">25</xref>] (2019)</td>
<td>&#x2013;/Alexnet CNN features/tandem of classifiers by differential evolution</td>
<td>Healthy = 62, grade 1 = 96, grade 2 = 99, grade 3 = 100</td>
<td>4</td>
<td>98.29</td>
<td>&#x2013;</td>
<td>0.9942</td>
<td>0.9840</td>
<td>&#x2013;</td>
</tr>
<tr>
<td/>
<td><bold>Proposed</bold></td>
<td>High level features (Inception-Resnet v2), Hand-crafted features/Hybrid Features/ensemble BO-RF</td>
<td>Healthy = 62, grade 1 = 96, grade 2 = 99, grade 3 = 100</td>
<td>4</td>
<td>99.16</td>
<td>0.9923</td>
<td>0.9971</td>
<td>0.9923</td>
<td>0.9894</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Comparative analysis of the proposed framework with existing techniques on the benchmark datasets, i.e., GlaS and IMEDIATREAT datasets, is illustrated in <xref ref-type="table" rid="table-6">Tab. 6</xref>. The proposed model is a four-class magnification-independent colon cancer grading framework evaluated on four different datasets with various magnifications; the accuracy of 97.58% and 99.16% were obtained for GlaS and IMEDIATREAT datasets, respectively. The performance of the proposed magnification-independent framework on GlaS dataset has surpassed previous studies [<xref ref-type="bibr" rid="ref-24">24</xref>] and [<xref ref-type="bibr" rid="ref-29">29</xref>]. However, the method presented in [<xref ref-type="bibr" rid="ref-17">17</xref>] exhibited slightly better accuracy (98.60%) than that of the proposed method because the segmentation performed is meant to work on specific magnified images (10X and 20X), and a three-class grading classification has been performed. Thus, the gland features extracted from these segmented regions are also dependent on segmentation outcomes, which are subsequently appropriate for a specific magnification and may not perform well for other low or high magnifications. Moreover, the proposed magnification-independent four-class grading framework shows better sensitivity (0.9807), specificity (0.9907), and MCC (0.9780) than the sensitivity (0.9730), specificity (0.9900), and MCC (0.9640) achieved in [<xref ref-type="bibr" rid="ref-17">17</xref>] and evaluating only the accuracy would be a biased decision. For the IMEDIATREAT dataset, the study presented in [<xref ref-type="bibr" rid="ref-30">30</xref>] developed segmentation with intensity-based thresholding, and morphological features were extracted for four-class grading to attain 89.75% accuracies; another study presented in [<xref ref-type="bibr" rid="ref-25">25</xref>] classified images into four-class using a tandem of classifiers with extracted deep CNN features on Alexnet and attained an accuracy of 98.29%. These accuracies are lesser than those acquired using the proposed method (99.16%) and evaluated under the same datasets. Sensitivity and F-score values of 0.9923 also show the proposed model&#x2019;s supremacy over existing techniques on the IMEDIATREAT dataset. Thus, the proposed magnification-independent colon cancer multiclass framework is a generalized framework over multiple datasets and magnifications.</p>
<p>As the images are from different image sources acquired through different staining conditions, the proposed method stain normalizes images, making them uniform across datasets. The proposed framework is modeled as a magnification-independent framework evaluated to work when trained respective or irrespective of magnifications and classifies any input samples as cross-training, with testing performed across magnifications. Thus, the proposed colon cancer grading method is an effective, generalized system with an average accuracy in the range of 94.40%&#x2013;99.16% across four different datasets from different country locations and various magnifications (4X, 10X, 20X, and 40X).</p>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Discussion</title>
<p>The proposed grading model demonstrated accurate four-class grading of colon cancer samples as an automated computational prototype. This research focuses on extracting various features, such as morphology, texture, and color for different colon image magnifications. The experimental analysis was conducted on various datasets, and the calculated outcome was satisfactory and superior to that presented in the literature. The proposed hybrid features are intended to extract all possible features for the four classes. The multi-feature-based classification method yielded better results than the individual-feature-based classification methods. Further, the proposed RF classifier hyperparameter was optimized using Bayesian optimization, which is more accurate than the traditional method. A one-vs-one strategy was adopted, ensuring an accurate outcome for multiclass classification to achieve consistent classification modeling for four-class grading. There are various advantages to our proposed system model over existing techniques. First, the proposed framework is a magnification-independent model that can work with any magnification of colon samples. Second, this algorithm requires no training and can be applied without a pre-trained model to any new specimen. Finally, the process does not require complex hardware and can be performed on desktop computers using any processor.</p>
<p>In particular, our approach has achieved great precision for the four-class colon cancer grading (IMEDIATREAT = 99.16% GlaS = 97.58%, IPC = 97.25% and AMC = 94.40%). Notably, the most discriminatory features emphasized by the proposed, containing cartoon features, Gabor features, color features, and morphological features, are the dominant features used to grade colon cancer samples&#x2019; malignancy. When the features were considered cumulative as a hybrid feature set, the model was less sensitive towards different magnifications and could grade the colon images more precisely.</p>
<p>The model was trained on a dataset from one source and tested on a dataset from another source to ensure that the proposed model was suitable for various data sources. Previous studies focused on the three-class grading of colon cancer [<xref ref-type="bibr" rid="ref-17">17</xref>,<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-29">29</xref>] for the GlaS dataset. The results from testing the proposed grading method (<xref ref-type="table" rid="table-2">Tabs. 2</xref>&#x2013;<xref ref-type="table" rid="table-5">5</xref>) support the four-class grading system and evidence the framework&#x2019;s efficiency. The proposed classification and feature combinations herein provide a novel, reliable categorization of colorectal cancer image datasets from various sources irrespective of magnifications. The proposed model performs for any dataset input image even if it is not included in the training sample. Our proposed technology assessment shows strongly that our model functions well in typical clinical contexts where dataset samples are more varied than in controlled laboratory environments. However, the proposed method lacks the precise geometric tabulation of the cells across different grades as it is meant to work on different magnifications. The imbalanced dataset images and noise variations in the images can deteriorate the performance of the model.</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Conclusion</title>
<p>The presented work proposes a magnification-independent colon cancer grading framework with a hybrid set of features, i.e., texture, color, and morphological features, and classifies images into four-class colon grades: normal, well, moderate, and poor. The proposed colon cancer grading framework includes a preprocessing phase comprising stain normalization, contrast enhancement, grayscale conversion, and K-means clustering to enhance the image quality and normalize the images across multiple datasets. The rich information regarding the image texture, edges, and structures across magnifications and grades are extracted from the texture features, including the cartoon features, Gabor wavelets, and wavelet moments. The color distribution across various grades was quantified with the color feature set comprising the HSV histogram, color auto-correlogram, and color moments. Morphological features extracted from the white cluster obtained through K-means clustering quantified the geometric variations across magnifications and grades. All extracted features were concatenated to create a rich, hybrid feature set for classification using majority voting on six Bayesian optimized RF classifiers. The experiments were conducted on four datasets with different magnification factors: IPC (4X, 10X, 40X), AMC (10X, 20X, 40X), GlaS (20X), and IMEDIATREAT (10X) to analyze the robustness of the proposed system model, wherein the IMEDIATREAT dataset calculated the highest accuracy of 99.16% followed by GlaS (97.58%), IPC (97.25%), and AMC (94.40%) datasets. Multiclass classification with optimized RF ensures the optimal accuracy of the proposed system. The proposed grading system was evaluated under various validation structures for generalizability and cross-training, and testing it as an independent model displayed promising results. In the future, magnification-independent segmentation can be implemented for grading and used to calculate and compare clinical results.</p>
</sec>
</body>
<back>
<ack><p>The authors thank Aster Medcity (Kochi, India) and Ishita Pathology Center (Allahabad, India) for the research&#x2019;s involvement by providing quality images that are applaudable. The continuous support of Dr. Ranjana Srivastava (Ishita Pathology Center), Dr. Sarah Kuruvila (Former Senior Consultant, Aster Medcity, Kerala), and Dr. Jyotima Agarwal is highly commendable. At the time of dataset image collection, Dr. Shahin Hameed was part of Aster Medcity and has extended his valuable support throughout the research.</p></ack>
<fn-group><fn fn-type="other"><p><bold>Funding Statement:</bold> This work was partially supported by the Research Groups Program (Research Group Number RG-1439-033), under the Deanship of Scientific Research, King Saud University, Riyadh, Saudi Arabia.</p></fn>
<fn fn-type="conflict"><p><bold>Conflicts of Interest:</bold> The authors declare that they have no conflicts of interest to report regarding the present study.</p></fn></fn-group>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><collab>WHO</collab></person-group>, &#x201C;<article-title>Cancer</article-title>.&#x201D; <year>Accessed 16 December 2020</year>, [Online]. Available at: <uri>https://www.who.int/news-room/fact-sheets/detail/cancer</uri>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Bray</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Ferlay</surname></string-name>, <string-name><given-names>I.</given-names> <surname>Soerjomataram</surname></string-name>, <string-name><given-names>R. L.</given-names> <surname>Siegel</surname></string-name>, <string-name><given-names>L. A.</given-names> <surname>Torre</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Global cancer statistics 2018: Globocan estimates of incidence and mortality worldwide for 36 cancers in 185 countries</article-title>,&#x201D; <source>CA: A Cancer Journal for Clinicians</source>, vol. <volume>68</volume>, no. <issue>6</issue>, pp. <fpage>394</fpage>&#x2013;<lpage>424</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Fleming</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ravula</surname></string-name>, <string-name><given-names>S. F.</given-names> <surname>Tatishchev</surname></string-name> and <string-name><given-names>H. L.</given-names> <surname>Wang</surname></string-name></person-group>, &#x201C;<article-title>Colorectal carcinoma: Pathologic aspects</article-title>,&#x201D; <source>Journal of Gastrointestinal Oncology</source>, vol. <volume>3</volume>, no. <issue>3</issue>, pp. <fpage>153</fpage>&#x2013;<lpage>173</lpage>, <year>2012</year>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W. K.</given-names> <surname>Blenkinsopp</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Stewart-Brown</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Blesovsky</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Kearney</surname></string-name> and <string-name><given-names>L. P.</given-names> <surname>Fielding</surname></string-name></person-group>, &#x201C;<article-title>Histopathology reporting in large bowel cancer</article-title>,&#x201D; <source>Journal of Clinical Pathology</source>, vol. <volume>34</volume>, no. <issue>5</issue>, pp. <fpage>509</fpage>&#x2013;<lpage>513</lpage>, <year>1981</year>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A. M.</given-names> <surname>Khan</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Rajpoot</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Treanor</surname></string-name> and <string-name><given-names>D.</given-names> <surname>Magee</surname></string-name></person-group>, &#x201C;<article-title>A nonlinear mapping approach to stain normalization in digital histopathology images using image-specific color deconvolution</article-title>,&#x201D; <source>IEEE Transactions on Biomedical Engineering</source>, vol. <volume>61</volume>, no. <issue>6</issue>, pp. <fpage>1729</fpage>&#x2013;<lpage>1738</lpage>, <year>2014</year>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Raghesh Krishnan</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Radhakrishnan</surname></string-name></person-group>, &#x201C;<article-title>Hybrid approach to classification of focal and diffused liver disorders using ultrasound images with wavelets and texture features</article-title>,&#x201D; <source>IET Image Processing</source>, vol. <volume>11</volume>, no. <issue>7</issue>, pp. <fpage>530</fpage>&#x2013;<lpage>538</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. N.</given-names> <surname>Gurcan</surname></string-name>, <string-name><given-names>L. E.</given-names> <surname>Boucheron</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Can</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Madabhushi</surname></string-name>, <string-name><given-names>N. M.</given-names> <surname>Rajpoot</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Histopathological image analysis: A review</article-title>,&#x201D; <source>IEEE Reviews in Biomedical Engineering</source>, vol. <volume>2</volume>, pp. <fpage>147</fpage>&#x2013;<lpage>171</lpage>, <year>2009</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>N.</given-names> <surname>Elazab</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Soliman</surname></string-name>, <string-name><given-names>S.</given-names> <surname>El-Sappagh</surname></string-name>, <string-name><given-names>S. M. R.</given-names> <surname>Islam</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Elmogy</surname></string-name></person-group>, &#x201C;<article-title>Objective diagnosis for histopathological images based on machine learning techniques: Classical approaches and new Trends</article-title>,&#x201D; <source>Mathematics</source>, vol. <volume>8</volume>, no. <issue>11</issue>, pp. <fpage>1863</fpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Rathore</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Hussain</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Khan</surname></string-name></person-group>, &#x201C;<article-title>Automated colon cancer detection using hybrid of novel geometric features and some traditional features</article-title>,&#x201D; <source>Computers in Biology and Medicine</source>, vol. <volume>65</volume>, pp. <fpage>279</fpage>&#x2013;<lpage>296</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Rathore</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Hussain</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Aksam Iftikhar</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Jalil</surname></string-name></person-group>, &#x201C;<article-title>Novel structural descriptors for automated colon cancer detection and grading</article-title>,&#x201D; <source>Computer Methods and Programs in Biomedicine</source>, vol. <volume>121</volume>, no. <issue>2</issue>, pp. <fpage>92</fpage>&#x2013;<lpage>108</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Rathore</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Aksam Iftikhar</surname></string-name></person-group>, &#x201C;<article-title>CBISC: A novel approach for colon biopsy image segmentation and classification</article-title>,&#x201D; <source>Arabian Journal for Science and Engineering</source>, vol. <volume>41</volume>, pp. <fpage>5061</fpage>&#x2013;<lpage>5076</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Babu</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Gupta</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Singh</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Hameed</surname></string-name></person-group>, &#x201C;<article-title>Colon cancer prediction on different magnified colon biopsy images</article-title>,&#x201D; in <conf-name>Proc. Tenth Int. Conf. on Advanced Computing</conf-name>, <publisher-loc>Chennai, India</publisher-loc>, pp. <fpage>277</fpage>&#x2013;<lpage>280</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Babu</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Gupta</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Singh</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Hameed</surname></string-name>, <string-name><given-names>R</given-names> <surname>Nayar</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Cancer screening on Indian colon biopsy images using texture and morphological features</article-title>,&#x201D; in <conf-name>Proc. Int. Conf. on Communication and Signal Processing</conf-name>, <publisher-loc>Chennai, India</publisher-loc>, pp. <fpage>0175</fpage>&#x2013;<lpage>0181</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Babu</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Singh</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Gupta</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Hameed</surname></string-name></person-group>, &#x201C;<article-title>Colon cancer detection in biopsy images for Indian population at different magnification factors using texture features</article-title>,&#x201D; in <conf-name>Proc. Ninth Int. Conf. on Advanced Computing</conf-name>, <publisher-loc>Chennai, India</publisher-loc>, pp. <fpage>192</fpage>&#x2013;<lpage>197</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Babu</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Gupta</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Singh</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Hameed</surname></string-name></person-group>, &#x201C;<article-title>Prediction of normal &#x0026; grades of cancer on colon biopsy images at different magnifications using minimal robust texture &#x0026; morphological features</article-title>,&#x201D; <source>Indian Journal of Public Health Research &#x0026; Development</source>, vol. <volume>11</volume>, no. <issue>1</issue>, pp. <fpage>695</fpage>&#x2013;<lpage>701</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>E.</given-names> <surname>Abdulhay</surname></string-name>, <string-name><given-names>M. A.</given-names> <surname>Mohammed</surname></string-name>, <string-name><given-names>D. A.</given-names> <surname>Ibrahim</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Arunkumar</surname></string-name> and <string-name><given-names>V.</given-names> <surname>Venkatraman</surname></string-name></person-group>, &#x201C;<article-title>Computer aided solution for automatic segmenting and measurements of blood leucocytes using static microscope images</article-title>,&#x201D; <source>Journal of Medical Systems</source>, vol. <volume>42</volume>, no. <issue>4</issue>, pp. <fpage>58</fpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Rathore</surname></string-name>, <string-name><given-names>M. A.</given-names> <surname>Iftikhar</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Chaddad</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Niazi</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Karasic</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Segmentation and grade prediction of colon cancer digital pathology images across multiple institutions</article-title>,&#x201D; <source>Cancers</source>, vol. <volume>11</volume>, no. <issue>11</issue>, pp. <fpage>1700</fpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Sirinukunwattana</surname></string-name>, <string-name><given-names>D. R. J.</given-names> <surname>Snead</surname></string-name> and <string-name><given-names>N. M.</given-names> <surname>Rajpoot</surname></string-name></person-group>, &#x201C;<article-title>A stochastic polygons model for glandular structures in colon histology images</article-title>,&#x201D; <source>IEEE Transactions on Medical Imaging</source>, vol. <volume>34</volume>, no. <issue>11</issue>, pp. <fpage>2366</fpage>&#x2013;<lpage>2378</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Husham</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Mustapha</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Mostafa</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Al-Obaidi</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Mohammed</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Comparative analysis between active contour and otsu thresholding segmentation algorithms in segmenting brain tumor magnetic resonance imaging</article-title>,&#x201D; <source>Journal of Information Technology Management</source>, vol. <volume>12</volume>, pp. <fpage>48</fpage>&#x2013;<lpage>61</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>I. J.</given-names> <surname>Hussein</surname></string-name>, <string-name><given-names>M. A.</given-names> <surname>Burhanuddin</surname></string-name>, <string-name><given-names>M. A.</given-names> <surname>Mohammed</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Elhoseny</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Garcia-Zapirain</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Fully automatic segmentation of gynaecological abnormality using a new viola-jones model</article-title>,&#x201D; <source>Computers, Materials &#x0026; Continua</source>, vol. <volume>66</volume>, no. <issue>3</issue>, pp. <fpage>3161</fpage>&#x2013;<lpage>3182</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Janowczyk</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Madabhushi</surname></string-name></person-group>, &#x201C;<article-title>Deep learning for digital pathology image analysis: A comprehensive tutorial with selected use cases</article-title>,&#x201D; <source>Journal of Pathology Informatics</source>, vol. <volume>7</volume>, pp. <fpage>29</fpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>P.</given-names> <surname>Kainz</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Pfeiffer</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Urschler</surname></string-name></person-group>, &#x201C;<article-title>Semantic segmentation of colon glands with deep convolutional neural networks and total variation segmentation</article-title>,&#x201D; <comment>ArXiv, arXiv: 1511.06919</comment>, vol. <volume>5</volume>, pp. <fpage>e3874</fpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Xu</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Jia</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Ai</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Zhang</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Large scale tissue histopathology image classification, segmentation, and visualization via deep convolutional activation features</article-title>,&#x201D; <source>BMC Bioinformatics</source>, vol. <volume>18</volume>, pp. <fpage>281</fpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Awan</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Sirinukunwattana</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Epstein</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Jefferyes</surname></string-name>, <string-name><given-names>U.</given-names> <surname>Qidwai</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Glandular morphometrics for objective grading of colorectal adenocarcinoma histology images</article-title>,&#x201D; <source>Scientific Reports</source>, vol. <volume>7</volume>, pp. <fpage>16852</fpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Lichtblau</surname></string-name> and <string-name><given-names>C.</given-names> <surname>Stoean</surname></string-name></person-group>, &#x201C;<article-title>Cancer diagnosis through a tandem of classifiers for digitized histopathological slides</article-title>,&#x201D; <source>PLOS ONE</source>, vol. <volume>14</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>20</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F. A.</given-names> <surname>Spanhol</surname></string-name>, <string-name><given-names>L. S.</given-names> <surname>Oliveira</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Petitjean</surname></string-name> and <string-name><given-names>L. A.</given-names> <surname>Heutte</surname></string-name></person-group>, &#x201C;<article-title>Dataset for breast cancer histopathological image classification</article-title>,&#x201D; <source>IEEE Transactions on Biomedical Engineering</source>, vol. <volume>63</volume>, no. <issue>7</issue>, pp. <fpage>1455</fpage>&#x2013;<lpage>1462</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>O.</given-names> <surname>Iizuka</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Kanavati</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Kato</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Rambeau</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Arihiro</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Deep learning models for histopathological classification of gastric and colonic epithelial tumours</article-title>,&#x201D; <source>Scientific Reports</source>, vol. <volume>10</volume>, pp. <fpage>12</fpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Kather</surname></string-name>, <string-name><given-names>C. A.</given-names> <surname>Weis</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Bianconi</surname></string-name>, <string-name><given-names>S. M.</given-names> <surname>Melchers</surname></string-name>, <string-name><given-names>L. R.</given-names> <surname>Schad</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Multi-class texture analysis in colorectal cancer histology</article-title>,&#x201D; <source>Scientific Reports</source>, vol. <volume>6</volume>, pp. <fpage>27988</fpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Saroja</surname></string-name> and <string-name><given-names>A. S.</given-names> <surname>Priyadharson</surname></string-name></person-group>, &#x201C;<article-title>Adaptive pillar k-means clustering-based colon cancer detection from biopsy samples with outliers</article-title>,&#x201D; <source>Computer Methods in Biomechanics and Biomedical Engineering Imaging &#x0026; Visualization</source>, vol. <volume>7</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>11</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Boruz</surname></string-name> and <string-name><given-names>C.</given-names> <surname>Stoean</surname></string-name></person-group>, &#x201C;<article-title>On supporting cancer grading based on histological slides using a limited number of features</article-title>,&#x201D; <source>Annals of the University of Craiova, Mathematics and Computer Science Series</source>, vol. <volume>45</volume>, no. <issue>1</issue>, pp. <fpage>156</fpage>&#x2013;<lpage>165</lpage>, <year>2018</year>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Stoean</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Stoean</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Sandita</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Ciobanu</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Mesina</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Svm-based cancer grading from histopathological images using morphological and topological features of glands and nuclei</article-title>,&#x201D; in <conf-name>Proc. Intelligent Interactive Multimedia Systems and Services</conf-name>, pp. <fpage>145</fpage>&#x2013;<lpage>155</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Nawandhar</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Kumar</surname></string-name>, <string-name><surname>V. R</surname></string-name> and <string-name><given-names>L.</given-names> <surname>Yamujala</surname></string-name></person-group>, &#x201C;<article-title>Stratified squamous epithelial biopsy image classifier using machine learning and neighborhood feature selection</article-title>,&#x201D; <source>Biomedical Signal Processing and Control</source>, vol. <volume>55</volume>, pp. <fpage>101671</fpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Rathore</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Niazi</surname></string-name>, <string-name><given-names>M. A.</given-names> <surname>Iftikhar</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Chaddad</surname></string-name></person-group>, &#x201C;<article-title>Glioma grading via analysis of digital pathology images using machine learning</article-title>,&#x201D; <source>Cancers</source>, vol. <volume>12</volume>, no. <issue>3</issue>, pp. <fpage>578</fpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Roy</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Lal</surname></string-name> and <string-name><given-names>J. R.</given-names> <surname>Kini</surname></string-name></person-group>, &#x201C;<article-title>Novel color normalization method for hematoxylin eosin stained histopathology images</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>7</volume>, pp. <fpage>28982</fpage>&#x2013;<lpage>28998</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H. O.</given-names> <surname>Lyon</surname></string-name>, <string-name><given-names>A. P.</given-names> <surname>De Leenheer</surname></string-name>, <string-name><given-names>R. W.</given-names> <surname>Horobin</surname></string-name>, <string-name><given-names>W. E.</given-names> <surname>Lambert</surname></string-name>, <string-name><given-names>E. K.</given-names> <surname>Schulte</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Standardization of reagents and methods used in cytological and histological practice with emphasis on dyes, stains and chromogenic reagents</article-title>,&#x201D; <source>The Histochemical Journal</source>, vol. <volume>26</volume>, no. <issue>7</issue>, pp. <fpage>533</fpage>&#x2013;<lpage>544</lpage>, <year>1994</year>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Rathore</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Hussain</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Aksam Iftikhar</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Jalil</surname></string-name></person-group>, &#x201C;<article-title>Ensemble classification of colon biopsy images based on information rich hybrid features</article-title>,&#x201D; <source>Computers in Biology and Medicine</source>, vol. <volume>47</volume>, pp. <fpage>76</fpage>&#x2013;<lpage>92</lpage>, <year>2014</year>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. B.</given-names> <surname>Amin</surname></string-name>, <string-name><given-names>F. L.</given-names> <surname>Greene</surname></string-name>, <string-name><given-names>S. B.</given-names> <surname>Edge</surname></string-name>, <string-name><given-names>C. C.</given-names> <surname>Compton</surname></string-name>, <string-name><given-names>J. E.</given-names> <surname>Gershenwald</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>The eighth edition AJCC cancer staging manual: continuing to build a bridge from a population-based to a more &#x201C;personalized&#x201D; approach to cancer staging</article-title>,&#x201D; <source>CA: A Cancer Journal for Clinicians</source>, vol. <volume>67</volume>, no. <issue>2</issue>, pp. <fpage>93</fpage>&#x2013;<lpage>99</lpage>, <year>2017</year>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Zuiderveld</surname></string-name></person-group>, <source>Contrast Limited Adaptive Histogram Equalization</source>. <publisher-loc>USA</publisher-loc>: <publisher-name>Academic Press Professional, Inc.</publisher-name>, pp. <fpage>474</fpage>&#x2013;<lpage>485</lpage>, <year>1994</year>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Fukunaga</surname></string-name> and <string-name><given-names>L.</given-names> <surname>Hostetler</surname></string-name></person-group>, &#x201C;<article-title>The estimation of the gradient of a density function, with applications in pattern recognition</article-title>,&#x201D; <source>IEEE Transactions on Information Theory</source>, vol. <volume>21</volume>, no. <issue>1</issue>, pp. <fpage>32</fpage>&#x2013;<lpage>40</lpage>, <year>1975</year>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Chang</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Feng</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Zhu</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>R.</given-names> <surname>He</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>CT and MRI image fusion based on multiscale decomposition method and hybrid approach</article-title>,&#x201D; <source>IET Image Processing</source>, vol. <volume>13</volume>, no. <issue>1</issue>, pp. <fpage>83</fpage>&#x2013;<lpage>88</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Thomas</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Vijayan</surname></string-name></person-group>, &#x201C;<article-title>Automated colon cancer detection using kernel sparse representation based classifier</article-title>,&#x201D; <source>International Journal of Engineering and Advanced Technology</source>, vol. <volume>4</volume>, no. <issue>6</issue>, pp. <fpage>317</fpage>&#x2013;<lpage>321</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Wimmer</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Tamaki</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Tischendorf</surname></string-name>, <string-name><given-names>M.</given-names> <surname>H&#x00E4;fner</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Yoshida</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Directional wavelet based features for colonic polyp classification</article-title>,&#x201D; <source>Medical Image Analysis</source>, vol. <volume>31</volume>, pp. <fpage>16</fpage>&#x2013;<lpage>36</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Kherfi</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Ziou</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Bernardi</surname></string-name></person-group>, &#x201C;<article-title>Combining positive and negative examples in relevance feedback for content-based image retrieval</article-title>,&#x201D; <source>Journal of Visual Communication and Image Representation</source>, vol. <volume>14</volume>, no. <issue>4</issue>, pp. <fpage>428</fpage>&#x2013;<lpage>457</lpage>, <year>2003</year>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Huang</surname></string-name>, <string-name><given-names>S. R.</given-names> <surname>Kumar</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Mitra</surname></string-name>, <string-name><given-names>W.-J.</given-names> <surname>Zhu</surname></string-name> and <string-name><given-names>R.</given-names> <surname>Zabih</surname></string-name></person-group>, &#x201C;<article-title>Image indexing using color correlograms</article-title>,&#x201D; in <conf-name>Proc. IEEE Computer Society Conf. on Computer Vision and Pattern Recognition</conf-name>, <publisher-loc>San Juan, Puerto Rico, USA</publisher-loc>, pp. <fpage>762</fpage>&#x2013;<lpage>768</lpage>, <year>1997</year>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>X.</given-names> <surname>Min</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Dong</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Feng</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Zhang</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Multi-parametric MRI-based radiomics signature for discriminating between clinically significant and insignificant prostate cancer: cross-validation of a machine learning method</article-title>,&#x201D; <source>European Journal of Radiology</source>, vol. <volume>115</volume>, no. <issue>6</issue>, pp. <fpage>16</fpage>&#x2013;<lpage>21</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-46"><label>[46]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Zahangir Alam</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Saifur Rahman</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Sohel Rahman</surname></string-name></person-group>, &#x201C;<article-title>A random forest based predictor for medical data classification using feature ranking</article-title>,&#x201D; <source>Informatics in Medicine Unlocked</source>, vol. <volume>15</volume>, <fpage>100180</fpage>, pp. <fpage>1</fpage>&#x2013;12, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-47"><label>[47]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Zakariah</surname></string-name></person-group>, &#x201C;<article-title>Classification of large datasets using random forest algorithm in various applications: survey</article-title>,&#x201D; <source>International Journal of Engineering and Innovative Technology</source>, vol. <volume>4</volume>, no. <issue>3</issue>, pp. <fpage>189</fpage>&#x2013;<lpage>198</lpage>, <year>2014</year>.</mixed-citation></ref>
<ref id="ref-48"><label>[48]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Trehan</surname></string-name></person-group>, &#x201C;<article-title>Why choose random forest and not decision trees</article-title>,&#x201D; <source>Towards AI&#x2014;Multidisciplinary Science Journal</source>, <year>2020</year>. [Online]. Available at: <uri>https://towardsai.net/p/machine-learning/why-choose-random-forest-and-not-decision-trees</uri>.</mixed-citation></ref>
<ref id="ref-49"><label>[49]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W. M.</given-names> <surname>Czarnecki</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Podlewska</surname></string-name> and <string-name><given-names>A. J.</given-names> <surname>Bojarski</surname></string-name></person-group>, &#x201C;<article-title>Robust optimization of SVM hyperparameters in the classification of bioactive compounds</article-title>,&#x201D; <source>Journal of Cheminformatics</source>, vol. <volume>7</volume>, no. <issue>38</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>15</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-50"><label>[50]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Wu</surname></string-name>, <string-name><given-names>X.-Y.</given-names> <surname>Chen</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>L.-D.</given-names> <surname>Xiong</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Lei</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Hyperparameter optimization for machine learning models based on bayesian optimization</article-title>,&#x201D; <source>Journal of Electronic Science and Technology</source>, vol. <volume>17</volume>, no. <issue>1</issue>, pp. <fpage>26</fpage>&#x2013;<lpage>40</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-51"><label>[51]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>N.</given-names> <surname>Meinshausen</surname></string-name></person-group>, &#x201C;<article-title>Quantile regression forests</article-title>,&#x201D; <source>Journal of Machine Learning Research</source>, vol. <volume>7</volume>, no. <issue>35</issue>, pp. <fpage>983</fpage>&#x2013;<lpage>999</lpage>, <year>2006</year>.</mixed-citation></ref>
<ref id="ref-52"><label>[52]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J. B.</given-names> <surname>Reitsma</surname></string-name>, <string-name><given-names>A. S.</given-names> <surname>Glas</surname></string-name>, <string-name><given-names>A. W.</given-names> <surname>Rutjes</surname></string-name>, <string-name><given-names>R. J.</given-names> <surname>Scholten</surname></string-name>, <string-name><given-names>P. M.</given-names> <surname>Bossuyt</surname></string-name> <etal>et al.</etal></person-group><italic>,</italic> &#x201C;<article-title>Bivariate analysis of sensitivity and specificity produces informative summary measures in diagnostic reviews</article-title>,&#x201D; <source>Journal of Clinical Epidemiology</source>, vol. <volume>58</volume>, no. <issue>10</issue>, pp. <fpage>982</fpage>&#x2013;<lpage>990</lpage>, <year>2005</year>.</mixed-citation></ref>
<ref id="ref-53"><label>[53]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Chougrad</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Zouaki</surname></string-name> and <string-name><given-names>O.</given-names> <surname>Alheyane</surname></string-name></person-group>, &#x201C;<article-title>Convolutional neural networks for breast cancer screening: Transfer learning with exponential decay</article-title>,&#x201D; <source>Computer Methods and Programs in Biomedicine</source>, vol. <volume>157</volume>, pp. <fpage>19</fpage>&#x2013;<lpage>30</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-54"><label>[54]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H. M.</given-names> <surname>Ahmad</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ghuffar</surname></string-name> and <string-name><given-names>K.</given-names> <surname>Khurshid</surname></string-name></person-group>, &#x201C;<article-title>Classification of breast cancer histology images using transfer learning</article-title>,&#x201D; in <conf-name>Proc. 16th Int. Bhurban Conf. on Applied Sciences and Technology</conf-name>, <publisher-loc>Islamabad, Pakistan</publisher-loc>, pp. <fpage>328</fpage>&#x2013;<lpage>332</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-55"><label>[55]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Christian</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ioffe</surname></string-name>, <string-name><given-names>V.</given-names> <surname>Vanhoucke</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Alemi</surname></string-name></person-group>, &#x201C;<article-title>Inception-v4, inception-resnet and the impact of residual connections on learning</article-title>,&#x201D; <source>in Proc. Thirty-first AAAI Conf. on Artificial Intelligence (AAAI&#x2019;17)</source>, <publisher-loc>California, USA</publisher-loc>: <publisher-name>San Francisco</publisher-name>, pp. <fpage>4278</fpage>&#x2013;<lpage>4284</lpage>, <year>2016</year>.</mixed-citation></ref>
</ref-list>
</back>
</article>