<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">48119</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2024.048119</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Time and Space Efficient Multi-Model Convolution Vision Transformer for Tomato Disease Detection from Leaf Images with Varied Backgrounds</article-title>
<alt-title alt-title-type="left-running-head">Time and Space Efficient Multi-Model Convolution Vision Transformer for Tomato Disease Detection from Leaf Images with Varied Backgrounds</alt-title>
<alt-title alt-title-type="right-running-head">Time and Space Efficient Multi-Model Convolution Vision Transformer for Tomato Disease Detection from Leaf Images with Varied Backgrounds</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Gangwar</surname><given-names>Ankita</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Dhaka</surname><given-names>Vijaypal Singh</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-3" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Rani</surname><given-names>Geeta</given-names></name><xref ref-type="aff" rid="aff-2">2</xref><email>geetechhikara@gmail.com</email></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Khandelwal</surname><given-names>Shrey</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Zumpano</surname><given-names>Ester</given-names></name><xref ref-type="aff" rid="aff-3">3</xref><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Vocaturo</surname><given-names>Eugenio</given-names></name><xref ref-type="aff" rid="aff-3">3</xref><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<aff id="aff-1"><label>1</label><institution>Department of Computer and Communication Engineering</institution>, <addr-line>Manipal University Jaipur</addr-line>, <addr-line>Jaipur</addr-line>, <country>India</country></aff>
<aff id="aff-2"><label>2</label><institution>Department of IoT and Intelligent Systems, Manipal University Jaipur</institution>, <addr-line>Jaipur</addr-line>, <country>India</country></aff>
<aff id="aff-3"><label>3</label><institution>Department of Computer Engineering, Modelling, Electronics and Systems (DIMES), University of Calabria</institution>, <addr-line>Rende (Cosenza)</addr-line>, <country>Italy</country></aff>
<aff id="aff-4"><label>4</label><institution>National Research Council, Institute of Nanotechnology (NANOTEC)</institution>, <addr-line>Rende (Cosenza)</addr-line>, <country>Italy</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Geeta Rani. Email: <email>geetechhikara@gmail.com</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic"><year>2024</year></pub-date>
<pub-date date-type="pub" publication-format="electronic"><day>25</day><month>4</month><year>2024</year></pub-date>
<volume>79</volume>
<issue>1</issue>
<fpage>117</fpage>
<lpage>142</lpage>
<history>
<date date-type="received"><day>28</day><month>11</month><year>2023</year></date>
<date date-type="accepted"><day>11</day><month>3</month><year>2024</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2024 Gangwar et al.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Gangwar et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_48119.pdf"></self-uri>
<abstract>
<p>A consumption of 46.9 million tons of processed tomatoes was reported in 2022 which is merely 20% of the total consumption. An increase of 3.3% in consumption is predicted from 2024 to 2032. Tomatoes are also rich in iron, potassium, antioxidant lycopene, vitamins A, C and K which are important for preventing cancer, and maintaining blood pressure and glucose levels. Thus, tomatoes are globally important due to their widespread usage and nutritional value. To face the high demand for tomatoes, it is mandatory to investigate the causes of crop loss and minimize them. Diseases are one of the major causes that adversely affect crop yield and degrade the quality of the tomato fruit. This leads to financial losses and affects the livelihood of farmers. Therefore, automatic disease detection at any stage of the tomato plant is a critical issue. Deep learning models introduced in the literature show promising results, but the models are difficult to implement on handheld devices such as mobile phones due to high computational costs and a large number of parameters. Also, most of the models proposed so far work efficiently for images with plain backgrounds where a clear demarcation exists between the background and leaf region. Moreover, the existing techniques lack in recognizing multiple diseases on the same leaf. To address these concerns, we introduce a customized deep learning-based convolution vision transformer model. The model achieves an accuracy of 93.51% for classifying tomato leaf images with plain as well as complex backgrounds into 13 categories. It requires a space storage of merely 5.8 MB which is 98.93%, 98.33%, and 92.64% less than state-of-the-art visual geometry group, vision transformers, and convolution vision transformer models, respectively. Its training time of 44 min is 51.12%, 74.12%, and 57.7% lower than the above-mentioned models. Thus, it can be deployed on (Internet of Things) IoT-enabled devices, drones, or mobile devices to assist farmers in the real-time monitoring of tomato crops. The periodic monitoring promotes timely action to prevent the spread of diseases and reduce crop loss.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Tomato</kwd>
<kwd>disease</kwd>
<kwd>transformer</kwd>
<kwd>deep learning</kwd>
<kwd>mobile devices</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>Department of Informatics, Modeling, Electronics and Systems (DIMES), University of Calabria</funding-source>
<award-id>SIMPATICO_ZUMPANO</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>According to data collected from the Food and Agriculture Organization Corporate Statistical Database, the world produced 189.1 million metric tonnes of tomatoes on 5,167,087 hectares in 2021 [<xref ref-type="bibr" rid="ref-1">1</xref>]. Average production is reported as 36.6 metric tonnes/hectare (mT/ha) [<xref ref-type="bibr" rid="ref-2">2</xref>]. As per data revealed by Business Standard News, a decline of approximately 4% was observed from 2019 to 2022 in the production of tomatoes [<xref ref-type="bibr" rid="ref-3">3</xref>,<xref ref-type="bibr" rid="ref-4">4</xref>].</p>
<p>Being a rich source of iron, potassium, antioxidant lycopene, and vitamins A, C and K, tomatoes are useful in preventing cancer, maintaining blood pressure, regulating blood glucose levels, and heart health. Thus, the consumption of tomatoes reached approximately 234.5 million metric tons in the year 2023. As for the statistics, 80% of tomatoes are consumed fresh and 20% are consumed as purees, soups, tomato ketchup, pickles, juices, sauces, etc. [<xref ref-type="bibr" rid="ref-5">5</xref>]. Moreover, an increase of 3.3% in demand for tomatoes is estimated from 2024 to 2032 [<xref ref-type="bibr" rid="ref-6">6</xref>].</p>
<p>Thus, a decline in production and an increase in demand become a driving force to investigate the causes of tomato crop loss and minimize the loss. Based on the literature, disease-prone nature, climate change, decrease in soil fertility, and lack of water availability are the major causes of tomato crop loss [<xref ref-type="bibr" rid="ref-7">7</xref>]. The susceptibility of tomato crops to diseases such as early blight, late blight, gray leaf mold, etc., is one of the leading factors for its crop loss. In the past decade, the maximum crop loss was observed due to the viral disease &#x2018;yellow leaf curl&#x2019; and the fungal disease &#x2018;late blight&#x2019; [<xref ref-type="bibr" rid="ref-8">8</xref>,<xref ref-type="bibr" rid="ref-9">9</xref>].</p>
<p>Diseases in tomato crops can be marked in the form of lesions on leaves, stems, blooms, and fruits of plants. The unique visual symptoms of each disease can be used for its detection [<xref ref-type="bibr" rid="ref-10">10</xref>]. Manual disease detection relies on global features such as texture, shape, the color of disease spotsx, etc. [<xref ref-type="bibr" rid="ref-11">11</xref>]. These methods are time-consuming and require expertise in disease identification [<xref ref-type="bibr" rid="ref-12">12</xref>]. Also, these techniques fail to predict crop loss based on disease severity. In recent years, some researchers have applied Deep Learning (DL) models to accurately detect diseases using the datasets collected from laboratories or fields [<xref ref-type="bibr" rid="ref-13">13</xref>&#x2013;<xref ref-type="bibr" rid="ref-16">16</xref>]. The DL models proposed by [<xref ref-type="bibr" rid="ref-7">7</xref>&#x2013;<xref ref-type="bibr" rid="ref-12">12</xref>] gave the highest accuracy of 99.35% in disease detection if the training and testing datasets were part of the same dataset. The models proposed by Mohanty et al. [<xref ref-type="bibr" rid="ref-14">14</xref>] and Ferentinos [<xref ref-type="bibr" rid="ref-15">15</xref>] reported a training accuracy of more than 99% but the performance was reduced to 35% when tested on an unseen dataset [<xref ref-type="bibr" rid="ref-13">13</xref>]. Now, Barbedo [<xref ref-type="bibr" rid="ref-16">16</xref>] highlighted asome factors influencing the performance of DL models applied to detect plant leaf diseases. They claimed that the proposed approaches still lack developing a generic tool for assisting farmers in real life.</p>
<p>Further advancements in DL techniques and the success of transformer neural networks in Natural Language Processing (NLP) tasks inspired the researchers to extend their applicability in image classification. Thus, the authors in [<xref ref-type="bibr" rid="ref-17">17</xref>] introduced Vision Transformers (ViT) for plant disease detection. The model exhibits slightly lower recognition accuracy than similar sized Convolutional Neural Networks (CNNs) trained on the same datasets. This decrease in accuracy can be attributed to two main reasons. Firstly, transformers process images as patches, leading to the omission of important local features like edges and lines, resulting in a loss of fine-grained details that CNNs captures well. Secondly, the attention layer in transformers may not be as efficient as the convolutional layers in CNNs for extracting fine details in terms of local features. To improve the performance of ViT model, Wu et al. proposed a hybrid model that incorporates a convolutional layer within the transformer architecture and introduced convolution to the ViT model. They proposed a Convolutional vision Transformer (CvT) model [<xref ref-type="bibr" rid="ref-18">18</xref>]. In a CNN, the convolution is performed on the input data using a filter to produce a feature map [<xref ref-type="bibr" rid="ref-19">19</xref>]. We perform element-wise multiplication of the input matrix and the filter element followed by calculating sum of the results to extract a feature. For example, Input matrix: <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mrow><mml:mo>[</mml:mo><mml:mtable columnalign="center center center" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>, and filter: <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mrow><mml:mo>[</mml:mo><mml:mtable columnalign="center center center" rowspacing="4pt" columnspacing="1em"><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>2</mml:mn></mml:mtd><mml:mtd><mml:mn>3</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>4</mml:mn></mml:mtd><mml:mtd><mml:mn>5</mml:mn></mml:mtd><mml:mtd><mml:mn>6</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>7</mml:mn></mml:mtd><mml:mtd><mml:mn>8</mml:mn></mml:mtd><mml:mtd><mml:mn>9</mml:mn></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula> are used to calculate the convolved feature as demonstrated below. The numbers in the input matrix and the filter that are multiplied together are represented with the same color.
<disp-formula id="ueqn-6"><mml:math id="mml-ueqn-6" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mrow><mml:mtext>Convolved</mml:mtext></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mtext>Feature</mml:mtext></mml:mrow></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>4</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>5</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>6</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>7</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>8</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd /><mml:mtd><mml:mi></mml:mi><mml:mspace width="1em" /><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>9</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>2</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>Similarly, all other values of the feature map are calculated as shown in the output feature map below.</p>
<p><inline-graphic xlink:href="CMC_48119-inline-1.tif"/></p>
<p>Convolution operation in the VGG16 model applies filters to extract hierarchical features, emphasizing spatial patterns. In VIT and CvT models, convolution is replaced by self-attention mechanisms, enabling global contextual understanding, and facilitating improved recognition of intricate disease-related details.</p>
<p>However, Deep learning (DL) and computer vision techniques have proven their potential in the automation of disease detection in tomato crops. Noisy background, identification of multiple diseases on the same plant or part of the plant, and varying symptoms of the same disease in different geographical regions create difficulty in disease identification [<xref ref-type="bibr" rid="ref-20">20</xref>]. Moreover, high computation costs, long training time, and the requirement of large storage space to deploy a DL model restrict their real-life implementation using handheld devices such as mobile phones, drones and IoT devices. Moreover, the size and quality of the dataset used for training the DL model highly affect its performance in disease detection [<xref ref-type="bibr" rid="ref-21">21</xref>]. This leaves room for improvement in existing models.</p>
<p>To address the above-mentioned challenges, we propose a tailored multi-model architecture &#x2018;Swift-Lite-CvT&#x2019; for disease detection in tomato plants to assist farmers and plant pathologists. The architecture is an integration of customized VGG16, ViT, and CvT models.</p>
<p>The key contributions of this study are listed below:
<list list-type="bullet">
<list-item>
<p>Preparing the labelled dataset including 13 tomato disease classes with plain, as well as, complex backgrounds.</p></list-item>
<list-item>
<p>Developing a multimodal &#x2018;Swift-Lite-CvT&#x2019; deep learning-based architecture for tomato disease detection.</p></list-item>
<list-item>
<p>Improving disease detection accuracy for datasets with complex backgrounds and multiple diseases on the same plant.</p></list-item>
<list-item>
<p>Reducing computation time and increasing the reliability of the DL-based architectures applied for tomato disease detection.</p></list-item>
</list></p>
<p>The remaining structure of the paper is as follows: <xref ref-type="sec" rid="s2">Section 2</xref> provides an overview of the related works. <xref ref-type="sec" rid="s3">Section 3</xref> presents the materials and methods including details about dataset preparation, and models employed for the experiment. <xref ref-type="sec" rid="s4">Section 4</xref> focuses on the experiments, <xref ref-type="sec" rid="s5">Section 5</xref> demonstrates the results obtained from trained models. <xref ref-type="sec" rid="s6">Section 6</xref> illustrates the discussion and provides a comprehensive analysis of the findings. The last section highlights the conclusions drawn and the future scope.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Works</title>
<p>In this section, we present the applied models, and the advantages and limitations observed in works available in literature. Thangaraj et al. [<xref ref-type="bibr" rid="ref-22">22</xref>] reviewed the challenges and limitations of the ML andDL models employed for tomato plant disease identification. Agarwal et al. [<xref ref-type="bibr" rid="ref-23">23</xref>] introduced a CNN model and evaluated its performance against established CNN models like VGG16, MobileNet, and InceptionV3. The proposed model achieved an accuracy of 91.20%, surpassing all the above-mentioned models. Similarly, Ahmad et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] collected a real field dataset comprising 317 images and applied augmentation techniques to increase the dataset size. Finally, they prepared an augmented dataset comprising 15,216 images. They applied VGG-16, VGG-19, ResNet, and Inception V3 models on the collected as well as augmented datasets. Inception V3 achieved the highest accuracy of 99.60%, and 93.70% on the augmented and collected dataset, respectively. Norria et al. [<xref ref-type="bibr" rid="ref-25">25</xref>] proposed an automated system for tomato diseases classification using ResNet50 CNN model. This system classifies the tomato leaf dataset into healthy, septoria leaf and late blight classes. The system reported an average accuracy of 92.08%. Next, Chowdhury et al. [<xref ref-type="bibr" rid="ref-26">26</xref>] used a plant village dataset for their studies. They applied ResNet18, MobilenetV2, InceptionV3, and DenseNet201 for binary classification, 6-class classification and 10-class classification. InceptionV3 outperformed and achieved an accuracy of 99.2% for binary classification. The DenseNet201 model attained 97.99% and 98.05% accuracy for 6-class and 10-class classification, respectively. This study has the potential for the early identification and automated diagnosis of diseases in tomato crops. Furthermore, by integrating a feedback system, this framework can offer valuable insights, treatments, preventive strategies, and disease control techniques, ultimately resulting in enhanced crop yields. In line with the previous works, Gonzalez-Huitron et al. [<xref ref-type="bibr" rid="ref-27">27</xref>] developed a GUI interface for the detection of disease in tomato leaves. For the experiments, they implemented lightweight CNN architectures on Raspberry Pi 4 microcomputer. Hassan et al. [<xref ref-type="bibr" rid="ref-28">28</xref>] developed an efficient DL model for disease detection in tomato leaves. They compared four DL models namely InceptionV3, InceptionResNetV2, MobileNetV2 and Efficient-NetB0 on the plant village dataset. EfficientNetB0, and MobileNet attained 99.56%, and 97.02% accuracy, respectively. However, MobileNet reports lower accuracy than EfficientNetB0, but its smaller number of parameters and lightweight architecture promote its real-life application using mobile devices. The model fails to handle noisy dataset, and complex background. Thus, leaves a scope for further research.</p>
<p>In all the above-discussed research works, a large training dataset is required. To address this issue, the authors in [<xref ref-type="bibr" rid="ref-29">29</xref>] applied a hybrid model developed using a Conditional Generative Adversarial Network (C-GAN) and a DL model. The model can generate synthetic images similar to real images and increase the dataset size. The authors applied DenseNet121 model for 5-class, 7-class and 10-class classification on the generated dataset and reported accuracy of 99.51%, 98.56% and 97.11%, respectively. Zhou et al. [<xref ref-type="bibr" rid="ref-30">30</xref>] used AI challenger dataset for tomato leaf disease detection using restructured residual dense network (RDN). This approach reduces the number of parameters, so it is an efficient model with less computation and attained an accuracy of 95%. Now, Paymode et al. [<xref ref-type="bibr" rid="ref-31">31</xref>] focused on disease detection in grapes and tomato crops. They collected real field datasets from Nashik, and Maharashtra, India for grape crops and used publicly available plant village dataset for tomato crops. They applied VGG16 model and achieved an accuracy of 95.71% for disease classification in tomatoes and 98.40% for grapes. The study deals with challenges in both collecting and preparing a genuine dataset, as well as the limited number of epochs used for training the data. Consequently, deploying this model on handheld devices with limited storage could pose challenges.</p>
<p>Similarly, Vadivel et al. [<xref ref-type="bibr" rid="ref-32">32</xref>] also applied VGG16 CNN model and gained a high accuracy of 99.5% on the plant village dataset. This study faces a challenge when tested on real-life datasets. Similarly, Tarek et al. [<xref ref-type="bibr" rid="ref-33">33</xref>] evaluated several DL models like Alex Net, ResNet50, InceptionV3, MobileNetV1, MobileNetV2 and MobileNetV3 on plant village datasets for tomato disease detection. MobileNetV3 model outperformed all the above-mentioned models and attained an accuracy of 99.81%. Moreover, the size of the model is reduced to 34 MB. Wang et al. [<xref ref-type="bibr" rid="ref-34">34</xref>] observed that existing classifiers face issues in recognizing diseases with similar symptoms, more than one disease on the same plant, and small disease lesions. Therefore, they collected the dataset from the three districts of Beijing, China and applied a combination of CNN models and transformer architecture for tomato disease detection. The model reported an accuracy of 96.30% on the real dataset. The authors in [<xref ref-type="bibr" rid="ref-35">35</xref>] used four publicly available datasets namely plant village [<xref ref-type="bibr" rid="ref-36">36</xref>], ibean [<xref ref-type="bibr" rid="ref-37">37</xref>], AI challenger [<xref ref-type="bibr" rid="ref-38">38</xref>] and PlantDoc [<xref ref-type="bibr" rid="ref-39">39</xref>] for plant disease detection. They applied Inception convolutional vision transformer model on the above-mentioned datasets and achieved an accuracy of 99.97%, 99.22%, 86.89% and 77.54%, respectively. But the model reported a low accuracy of 77.54% when applied to dataset with a complex background. Now, Alzahrani et al. [<xref ref-type="bibr" rid="ref-40">40</xref>] compared the performance of DenseNet169, ResNet50V2 and ViT models on the publicly available dataset [<xref ref-type="bibr" rid="ref-13">13</xref>] comprising leaf images with a plain background. DenseNet121 achieved the highest accuracy of 99%. Also, the models are intense and cannot be implemented on handheld devices. And, the model fails to deal with real-world data. The most related works and their limitations are summarized in <xref ref-type="table" rid="table-1">Table 1</xref>.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Summary of related works</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th>References</th>
<th>Number of classes</th>
<th>Contributions</th>
<th>Accuracy</th>
<th>Limitations</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td>Agarwal et al. 2020 [<xref ref-type="bibr" rid="ref-23">23</xref>]</td>
<td>10</td>
<td>Proposed CNN model has a storage space requirement of 1.5 MB.</td>
<td>91.2%</td>
<td rowspan="2">Scope to improve accuracy. Not evaluated for dataset with complex background.</td>
</tr>
<tr>
<td>Norria et al. 2021 [<xref ref-type="bibr" rid="ref-25">25</xref>]</td>
<td>3</td>
<td>Pre-processing, segmentation, feature extraction followed by CNN for tomato leaf disease classification.</td>
<td>92.08%</td>
</tr>
<tr>
<td>Hassan et al. 2021 [<xref ref-type="bibr" rid="ref-28">28</xref>]</td>
<td>38</td>
<td>Replaced standard convolution with depth separable convolution to reduce number of parameters and computation cost in EfficientNetB0 models.</td>
<td>99.56%</td>
<td>Lacks in handling noisy dataset, and complex background.</td>
</tr>
<tr>
<td>Paymode et al. 2022 [<xref ref-type="bibr" rid="ref-31">31</xref>]</td>
<td>10</td>
<td>Focused on the early detection, classification, and analysis of diseases, particularly in tomatoes and grapes, to aid agricultural progress. Implemented VGG16 model, for Multi Crops Leaf Disease (MCLD) detection.</td>
<td>95.71% in tomatoes and 98.40% in grapes</td>
<td>Scope to validate the results on dataset with complex background.</td>
</tr>
<tr>
<td>Wang et al. 2022 [<xref ref-type="bibr" rid="ref-34">34</xref>]</td>
<td>7</td>
<td>Improved the efficiency of feature extraction in plant images using a multi-grained model based on vision transformer and useful for the small training dataset.</td>
<td>96.30%</td>
<td>The model is space intensive. Also, there is scope to evaluate on large dataset.</td>
</tr>
<tr>
<td>Yu et al. 2022 [<xref ref-type="bibr" rid="ref-35">35</xref>]</td>
<td>27</td>
<td>Proposed a hybrid model using CNN and transformer architecture. Utilized soft split token embedding to capture local information from surrounding pixels and patches, enhancing fine grained feature learning.</td>
<td>77.54%</td>
<td>Lacks in handling variations in disease symptoms and environmental factors. Scope to improve the accuracy.</td>
</tr>
<tr>
<td>Alzahrani et al. 2023 [<xref ref-type="bibr" rid="ref-40">40</xref>]</td>
<td>10</td>
<td>Provided a plain and low cost DenseNet121 for diagnosing tomato leaf diseases only when user take an image of the affected plant&#x2019;s leaf.</td>
<td>99%</td>
<td>Scope to evaluate images with complex background. Need to reduce the space and computation time to make it feasible for real-life applications and to integrate with mobile applications.</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>According to the above discussion, it is evident that the employed DL models are less efficient in disease prediction from the dataset with complex backgrounds. Moreover, the models proposed so far have higher computational and memory requirements [<xref ref-type="bibr" rid="ref-17">17</xref>]. Thus, there is a need for an architecture that can correctly detect disease from the plain as well as complex background. Simultaneously, it should be small so that it can be implemented on handheld devices to assist farmers in automatic crop monitoring and disease prediction. To meet these challenges, we propose a multimodal &#x2018;Swift-Lite-CvT&#x2019; that correctly performs multi-class classification of the dataset comprising tomato leaf images with plain and complex backgrounds. The convolution operations performed by this model are efficient in capturing local spatial details, whereas its transformer part is effective in analyzing the global context of an image. Therefore, using the insights from the local and global features, the model can classify the image to the correct class irrespective of type of the background. Moreover, we focus on minimizing the computation and storage requirements of the model.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Materials and Methods</title>
<p>In this research, we applied various DL techniques to detect tomato leaf diseases and classify them into 10, 11 and 13 classes. For correct multi-class classification, we employed pre-trained networks, namely VGG16, Vision Transformer, and CvT models. Pretraining on large datasets of tomato leaves acts as an initialization, allowing the model to start with already optimized parameters. While pretraining, the model learns to recognize the shape, size, boundaries, etc. Further training of the pre-trained model on the labelled dataset and its fine tuning helps in faster converging, making the training process more computationally efficient. To further improve the reliability, and reduce the training time we designed a customized model &#x201C;Swift-Lite-CvT&#x201D;, based on CVT model [<xref ref-type="bibr" rid="ref-41">41</xref>,<xref ref-type="bibr" rid="ref-42">42</xref>]. The depth of transformer layer in the proposed model is reduced from 2 to 1 in second block, and 10 to 1 in the third block. Additionally, the number of attention heads are reduced from 3 to 2, and from 6 to 4 in the second and third blocks, respectively. These changes make the proposed model more lightweight and computationally efficient than the original CvT. The classification strategy followed in this study is illustrated in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Classification strategy: Models implemented for multi class classification</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48119-fig-1.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Datasets</title>
<p>In this study, three distinct datasets were employed, namely Plant Village [<xref ref-type="bibr" rid="ref-43">43</xref>], Taiwan dataset [<xref ref-type="bibr" rid="ref-44">44</xref>], and PlantDoc dataset [<xref ref-type="bibr" rid="ref-45">45</xref>] comprising tomato leaves. The dataset comprises tomato leaf images with a plain grey background and complex background having rotten leaves, stones, soil, dry leaves, etc. The dataset with complex background is prepared to train the model for real-life datasets captured directly from fields. The dataset captured from fields cannot have a plain background, thus, to enable the model for classifying real-life images to the correct disease or healthy class, the dataset with a complex background is prepared.</p>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Plant Village Dataset</title>
<p>The Plant village dataset [<xref ref-type="bibr" rid="ref-43">43</xref>] comprises 50,000 images of tomato leaves labelled with ten disease classes such as bacterial spot, early blight, healthy, late blight, leaf mold, septoria leaf spot, target spot, tomato mosaic virus, tomato yellow leaf curl virus and two-spotted spider mite. Each image contains a single leaf and a plain background. A plain background means that there is a clear demarcation between the leaf and the background region. The background region is uncoloured without any object. Sample images for this dataset are shown in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>. This dataset is divided into training and testing datasets in the ratio of 80:20, respectively.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Sample images from plant village dataset with plain background [<xref ref-type="bibr" rid="ref-43">43</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48119-fig-2.tif"/>
</fig>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Taiwan Dataset</title>
<p>The Taiwan dataset of tomato leaves [<xref ref-type="bibr" rid="ref-44">44</xref>] comprised images labelled with six categories such as bacterial spot, black leaf mold, gray leaf spot, healthy, late blight and powdery mildew. This dataset contains images with a single leaf, multiple leaves, a plain background and a complex background. A complex background contains one or more objects. Thus, there is no clear demarcation between the leaf and the background region. The sample images are shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>. The dataset comprises 622 original images. The size of images varies, so we unified them to 224 &#x00D7; 224. We also applied data augmentation techniques such as centre cropping, random cropping, horizontal flipping, etc., to increase the dataset size. The details of the dataset are shown in <xref ref-type="table" rid="table-2">Table 2</xref>.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Sample images of Taiwan dataset of tomato leaves with complex background [<xref ref-type="bibr" rid="ref-44">44</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48119-fig-3.tif"/>
</fig><table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Details of customized dataset containing tomato leaf images with plain and complex background</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th>Number of classes</th>
<th>Total number of images</th>
<th>Images in training dataset</th>
<th>Images in testing dataset</th>
<th>Images in validation dataset</th>
</tr>
</thead>
<tbody>
<tr>
<td>10 class</td>
<td>50,000</td>
<td>39,000</td>
<td>1,000</td>
<td>10,000</td>
</tr>
<tr>
<td>11 class</td>
<td>14,344</td>
<td>11,044</td>
<td>1,100</td>
<td>2,200</td>
</tr>
<tr>
<td>13 class</td>
<td>8,292</td>
<td>5,564</td>
<td>1,324</td>
<td>1,404</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_1_3">
<label>3.1.3</label>
<title>PlantDoc</title>
<p>The PlantDoc dataset [<xref ref-type="bibr" rid="ref-45">45</xref>] comprising 2,598 images was labelled with 13 plant species and 17 classes of diseased and healthy leaves. Also, the dataset contains human hands in the background as well as other plant parts forming the complex background. Sample images of this dataset are shown in <xref ref-type="fig" rid="fig-4">Fig. 4</xref>.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Sample images from PlantDoc dataset with complex background [<xref ref-type="bibr" rid="ref-45">45</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48119-fig-4.tif"/>
</fig>
</sec>
<sec id="s3_1_4">
<label>3.1.4</label>
<title>Customized Tomato Leaf Dataset (CTL Dataset)</title>
<p>To extend the applicability of the designed DL model &#x201C;Swift-Lite-CvT&#x201D;, we collected images from all the three above-mentioned datasets. Using solely plain background images proved insufficient for achieving accurate detection and classification of tomato plant leaf diseases in real life. So, we included images with plain and complex backgrounds. This enables better training of the model on the versatile dataset and improves the robustness of the model. Here, robustness is the potential of the model to accurately classify tomato leaf images captured with plain as well as complex backgrounds. Thus, the performance of the model remains consistent in data collected in a laboratory or a field. <xref ref-type="table" rid="table-2">Table 2</xref> shows the number of classes, dataset size for training, testing and validation of the model.</p>

</sec>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>DL Models</title>
<p>In this study, we have deployed the following four DL models, based on deep CNN and transformer architectures.</p>
<sec id="s3_2_1">
<label>3.2.1</label>
<title>VGG16</title>
<p>VGG16 [<xref ref-type="bibr" rid="ref-46">46</xref>] model used for plant leaf disease detection. The filter sizes were reduced to 11 and 5 in the first and second convolutional layers, respectively. Also, the large-sized kernel filters were replaced with multiple 3 &#x00D7; 3 kernel sized filters. The modification enables the model to capture more localized features within the images. To enhance the performance and generalization of the VGG16 model, augmentation techniques such as rotation, scaling, and flipping, were applied to the input images. The variation caused by augmenting the training data helps to improve the robustness of the model. Hence, the model becomes more efficient in classifying various plant diseases. Our selection of VGG16 was influenced by its compact architecture with a reduced number of layers and parameters. Its proven reliability and high accuracy of 99.53%, 95.2% and 89% for tomato disease detection demonstrated in [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-47">47</xref>,<xref ref-type="bibr" rid="ref-48">48</xref>] respectively further supported our choice.</p>
</sec>
<sec id="s3_2_2">
<label>3.2.2</label>
<title>Vision Transformer (ViT)</title>
<p>The Vision Transformer (ViT) [<xref ref-type="bibr" rid="ref-49">49</xref>] has emerged as a prominent neural network architecture in the field of computer vision. In this model, the convolution neural network component is replaced with the transformer block which was earlier designed for handling sequential data. Thus, ViT adapted to handle images by dividing it into a sequence of patches. It considers each patch as a token of the sequence and learns to correctly classify images with plain or complex backgrounds.</p>
<p>In this architecture, initially, the input image is divided into patches of equal size, as depicted in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>. Each patch is then flattened into a 1-D array vector and subsequently embedded with positional encoding. This encoding step helps to capture the spatial information of each patch within the image. The encoded patches are passed through the transformer encoder block. This block consists of self-attention mechanisms and Multi-Layer Perceptron (MLP) heads, which enable the model to capture both local and global dependencies within the image. Finally, classification is performed using the transformer architecture, leveraging the learned representations from the encoded patches. The ViT architecture divides the input image into patches, encoding with positional information, and then processing them through transformer encoder blocks for accurate image classification. The potential of ViT in extracting local as well as global features from an image and maintaining spatial relationships within an image motivated us to employ this model for tomato disease classification.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Input image in form of patches</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48119-fig-5.tif"/>
</fig>
</sec>
<sec id="s3_2_3">
<label>3.2.3</label>
<title>Convolutional Vision Transformer (CvT)</title>
<p>The CvT [<xref ref-type="bibr" rid="ref-50">50</xref>] architecture represents an improved version of the Vision Transformer that enhances performance and efficiency by incorporating convolutions. It combines the advantages of both CNNs and transformers. It adopts a hierarchical, multi-stage architecture, enabling progressive feature extraction and refinement. Its basic architecture is shown in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Convolution vision transformer model for tomato disease detection</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48119-fig-6.tif"/>
</fig>
<p>CvT model initiates its task with tokenization, followed by transformation and classification. The steps followed are illustrated below: Initially, a convolutional layer is employed to transform the image&#x2019;s local features into tokens, which are subsequently processed by the transformer block and normalized through layer normalization. Now, the transformer blocks capture extended-range dependencies and establish an interaction among tokens. These transformers incorporate multi-head attention. Each stage employs a unique number of heads. For example, number of heads are 1, 3, and 6 across three stages. This variation facilitates adaptable and dynamic receptive fields. Upon completion of processing through all stages, a classification token (cls token) is introduced at the outset, and the token representations are aggregated for the classification task. A MLP is then applied to this pooled representation for the final classification.</p>
<p>The experimental results presented in [<xref ref-type="bibr" rid="ref-14">14</xref>,<xref ref-type="bibr" rid="ref-27">27</xref>,<xref ref-type="bibr" rid="ref-50">50</xref>] validate that CvT outperforms both the Vision Transformer and ResNet model. Also, the removal of the positional encoding step, which is typically used in transformers to transform input images into patches reduces the training time of the model. Despite removing positional encoding, the CvT model maintains its performance. The CvT architecture leverages the advantages of convolutions to efficiently process input images while eliminating the need for explicit positional encoding. This motivated us to select CvT model for our research.</p>
</sec>
<sec id="s3_2_4">
<label>3.2.4</label>
<title>Modified Convolutional Vision Transfer (Swift-Lite-CvT)</title>
<p>We modified the architecture of the CvT model to improve its efficacy for tomato disease detection. The architecture of &#x201C;Swift-Lite-CvT&#x201D; is shown in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>. The modified version of CvT &#x201C;Swift-Lite-CvT&#x201D;, retains the hierarchical structure of the original CvT to preserve its proficiency in extracting local features. But the number of attention heads in the transformers is modified to 1, 2, and 4 across the three stages. Also, the depth of the transformer blocks is simplified and set 1 for all stages. Like the original CvT, after processing the tokens are pooled and subsequently channeled through a MLP for the classification task.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>(a) Overall architecture of proposed Swift-Lite-CvT architecture. (b) Details of Conv_Trans block</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48119-fig-7.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiments</title>
<p>In this section, the details of the experimental setup, implementation details, and evaluation metrics employed are discussed.</p>
<sec id="s4_1">
<label>4.1</label>
<title>Experimental Setup</title>
<p>For the experiments, a workstation with i9 processor, RTX 3070 GPU, 64 GB RAM and 3 TB hard disk is used. The pytorch and transformer architecture library are used for developing DL models implemented in this research.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Implementation</title>
<p>In the first block, &#x2018;Conv_trans1&#x2019;, convolutional token embedding is done with a 2D convolution, utilizing a kernel size of 7 and a stride of 4. The output channels generated by this convolution are referred to dim with a default value of 64. Now, the feature map obtained is transformed into tokens. These tokens are normalized through LayerNorm. Transformer block comprises a single transformer layer and the multi-head attention mechanism utilizes heads [0]. The token map generated is designated as &#x00D7;1. In the second block, &#x201C;Conv_trans2&#x201D;, convolutional token embedding employs a 2D convolution with a kernel size of 3 and a stride of 2. The number of output channels is determined as a scaled-up version of the dim from the previous stage. This scaling factor depends on the ratio of the attention heads in this stage to the previous stage, specifically heads [1]/heads [0]. Similar to block 1, the convolution&#x2019;s output is restructured into tokens and subsequently normalized. The transformer block contains a solitary transformer layer. In this layer, the multi head attention mechanism employs heads [1], which are configured to use two attention heads. It generates a token map denoted as &#x00D7;2. Next, in the third block Conv_trans3, works similar to Conv_trans2 except the ratio of heads is calculated as [2]/heads [1], and its multi-head attention mechanism utilizes heads [2]. Also, the multi-head attention has been configured to employ attention heads [<xref ref-type="bibr" rid="ref-4">4</xref>] in the modified model. Before it is passed through the transformer, a class (cls) token is added at the beginning of the token map. After processing through the transformer, the output tokens can be aggregated through either means pooling or by utilizing the cls token. Then, a normalization layer is applied followed by a linear layer responsible for reducing the feature dimension to be equal to the number of classes (num_classes). The model parameters improve reliability, and accuracy of disease detection. To further improve the performance of the model, a set of experiments are performed to finetune the hyperparameters of DL models. The fine-tuned hyperparameters for all the above-mentioned models applied for the 10, 11, and 13 class classifications are shown in <xref ref-type="table" rid="table-3">Table 3</xref>. The proposed model is trained for 100 epochs with a batch size of 16. The optimization process utilizes the stochastic gradient descent (SGD) optimizer in combination with the cross-entropy loss function. The cross entropy loss function is employed due to its effectiveness in multi-class classification. This loss function helps in the extraction of more distinctive features, thereby enhancing the process of making informed decisions and accurate predictions.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Hyperparameters for the models applied for the tomato disease detection</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th align="center" colspan="7">10 class classification</th>
</tr>
</thead>
<tbody>
<tr>
<td>Models</td>
<td>Momentum</td>
<td>Optimizer</td>
<td>Learning rate</td>
<td>Weight decay</td>
<td>Activation funtion</td>
<td>Patch size</td>
</tr>
<tr>
<td>VGG16</td>
<td>0.9</td>
<td>SGD</td>
<td>0.0005</td>
<td>0.0005</td>
<td>ReLu</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>ViT</td>
<td>0.9</td>
<td>SGD</td>
<td>0.01</td>
<td>&#x2013;</td>
<td>GeLu</td>
<td>16</td>
</tr>
<tr>
<td>CvT</td>
<td>0.9</td>
<td>SGD</td>
<td>0.01</td>
<td>&#x2013;</td>
<td>GeLu</td>
<td>7</td>
</tr>
<tr>
<td align="center" colspan="7">11 class classification</td>
</tr>
<tr>
<td>VGG16</td>
<td>0.9</td>
<td>SGD</td>
<td>0.0005</td>
<td>0.0005</td>
<td>ReLu</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>ViT</td>
<td>0.9</td>
<td>SGD</td>
<td>0.01</td>
<td>&#x2013;</td>
<td>GeLu</td>
<td>16</td>
</tr>
<tr>
<td>CvT</td>
<td>0.9</td>
<td>SGD</td>
<td>0.01</td>
<td>&#x2013;</td>
<td>GeLu</td>
<td>7</td>
</tr>
<tr>
<td align="center" colspan="7">13 class classification</td>
</tr>
<tr>
<td>VGG16</td>
<td>0.9</td>
<td>SGD</td>
<td>0.0005</td>
<td>0.0005</td>
<td>ReLu</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>ViT</td>
<td>0.9</td>
<td>SGD</td>
<td>0.01</td>
<td>&#x2013;</td>
<td>GeLu</td>
<td>16</td>
</tr>
<tr>
<td>CvT</td>
<td>0.9</td>
<td>SGD</td>
<td>0.01</td>
<td>&#x2013;</td>
<td>GeLu</td>
<td>7</td>
</tr>
<tr>
<td>Swift-Lite-CvT</td>
<td>0.9</td>
<td>SGD</td>
<td>0.01</td>
<td>&#x2013;</td>
<td>Gelu</td>
<td>7</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Evaluation Metrics</title>
<p>We employed confusion matrix, precision, recall, F1 score and classification accuracy to assess the performance of the models implemented in this study. These metrics are defined from <xref ref-type="disp-formula" rid="eqn-1">Eqs. (1)</xref>&#x2013;<xref ref-type="disp-formula" rid="eqn-6">(6)</xref>.</p>
<sec id="s4_3_1">
<label>4.3.1</label>
<title>Confusion Matrix</title>
<p>A confusion matrix contains the actual labels and the predicted labels for each class. For plain representation, we abbreviated tomato disease classes in <xref ref-type="table" rid="table-4">Table 4</xref>. The sample confusion matrix is shown in <xref ref-type="table" rid="table-5">Table 5</xref>. Here, T denotes true which indicates the number of correct classifications. F denotes false which indicates the number of incorrect classifications. For example, TBM is the number of correctly classified instances for class black mold. FBMG is the number of incorrectly classified instances of black mold to gray spot, and FBMBS is the number of incorrectly classified instances of black mold to bacterial spot. A similar notation is followed for all the correctly and incorrectly classified instances.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Class labels and their abbreviations</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th>Class label</th>
<th>Abbreviation</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td>Black mold</td>
<td>BM</td>
</tr>
<tr>
<td>Gray spot</td>
<td>G</td>
</tr>
<tr>
<td>Bacterial spot</td>
<td>BS</td>
</tr>
<tr>
<td>Early blight</td>
<td>E</td>
</tr>
<tr>
<td>Healthy</td>
<td>H</td>
</tr>
<tr>
<td>Late blight</td>
<td>LB</td>
</tr>
<tr>
<td>Leaf mold</td>
<td>LM</td>
</tr>
<tr>
<td>Mosaic virus</td>
<td>M</td>
</tr>
<tr>
<td>Powdery mildew</td>
<td>P</td>
</tr>
<tr>
<td>Septoria leaf spot</td>
<td>S</td>
</tr>
<tr>
<td>Two spotted spider mite</td>
<td>T</td>
</tr>
<tr>
<td>Target spot</td>
<td>TS</td>
</tr>
<tr>
<td>Yellow leaf curl virus</td>
<td>Y</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Sample confusion matrix for multi class classification</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th align="center" colspan="15">Predicted class</th>
</tr>
<tr>
<td></td>
<td></td>
<td>BM</td>
<td>G</td>
<td>BS</td>
<td>E</td>
<td>H</td>
<td>LB</td>
<td>LM</td>
<td>M</td>
<td>P</td>
<td>S</td>
<td><bold>T</bold></td>
<td>TS</td>
<td><bold>Y</bold></td>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="13">Actual class</td>
<td>BM</td>
<td>TBM</td>
<td>FBMG</td>
<td>FBMBS</td>
<td>FBME</td>
<td>FBMH</td>
<td>FBMLB</td>
<td>FBMLM</td>
<td>FBMM</td>
<td>FBMP</td>
<td>FBMS</td>
<td>FBMT</td>
<td>FBMTS</td>
<td>FBMY</td>
</tr>
<tr>
<td>G</td>
<td>FGBM</td>
<td>TG</td>
<td>FGBS</td>
<td>FGE</td>
<td>FGH</td>
<td>FGLB</td>
<td>FGLM</td>
<td>FGM</td>
<td>FGP</td>
<td>FGS</td>
<td>FGT</td>
<td>FGTS</td>
<td>FGY</td>
</tr>
<tr>
<td>BS</td>
<td>FBSBM</td>
<td>FBSG</td>
<td>TBS</td>
<td>FBSE</td>
<td>FBSH</td>
<td>FBSLB</td>
<td>FBSLM</td>
<td>FBSM</td>
<td>FBSP</td>
<td>FBSS</td>
<td>FBST</td>
<td>FBSTS</td>
<td>FBSY</td>
</tr>
<tr>
<td>E</td>
<td>FEBM</td>
<td>FEG</td>
<td>FEBS</td>
<td>TE</td>
<td>FEH</td>
<td>FELB</td>
<td>FELM</td>
<td>FEM</td>
<td>FEP</td>
<td>FES</td>
<td>FET</td>
<td>FETS</td>
<td>FEY</td>
</tr>
<tr>
<td>H</td>
<td>FHBM</td>
<td>FHG</td>
<td>FHBS</td>
<td>FHE</td>
<td>TH</td>
<td>FHLB</td>
<td>FHLM</td>
<td>FHM</td>
<td>FHP</td>
<td>FHS</td>
<td>FHT</td>
<td>FHTS</td>
<td>FHY</td>
</tr>
<tr>
<td>LB</td>
<td>FLBBM</td>
<td>FLBG</td>
<td>FLBBS</td>
<td>FLBE</td>
<td>FLBH</td>
<td>TLB</td>
<td>FLBLM</td>
<td>FLBM</td>
<td>FLBP</td>
<td>FLBS</td>
<td>FLBT</td>
<td>FLBTS</td>
<td>FLBY</td>
</tr>
<tr>
<td>LM</td>
<td>FLMBM</td>
<td>FLMG</td>
<td>FLMBS</td>
<td>FLME</td>
<td>FLMH</td>
<td>FLMLB</td>
<td>TLM</td>
<td>FLMM</td>
<td>FLMP</td>
<td>FLMS</td>
<td>FLMT</td>
<td>FLMTS</td>
<td>FLMY</td>
</tr>
<tr>
<td>M</td>
<td>FMBM</td>
<td>FMG</td>
<td>FMBS</td>
<td>FME</td>
<td>FMH</td>
<td>FMLB</td>
<td>FMLM</td>
<td>TM</td>
<td>FMP</td>
<td>FMS</td>
<td>FMT</td>
<td>FMTS</td>
<td>FMY</td>
</tr>
<tr>
<td>P</td>
<td>FPBM</td>
<td>FPG</td>
<td>FPBS</td>
<td>FPE</td>
<td>FPH</td>
<td>FPLB</td>
<td>FPLM</td>
<td>FPM</td>
<td>TP</td>
<td>FPS</td>
<td>FPT</td>
<td>FPTS</td>
<td>FPY</td>
</tr>
<tr>
<td>S</td>
<td>FSBM</td>
<td>FSG</td>
<td>FSBS</td>
<td>FSE</td>
<td>FSH</td>
<td>FSLB</td>
<td>FSLM</td>
<td>FSM</td>
<td>FSP</td>
<td>TS</td>
<td>FST</td>
<td>FSTS</td>
<td>FSY</td>
</tr>
<tr>
<td>T</td>
<td>FTBM</td>
<td>FTG</td>
<td>FTBS</td>
<td>FTE</td>
<td>FTH</td>
<td>FTLB</td>
<td>FTLM</td>
<td>FTM</td>
<td>FTP</td>
<td>FTS</td>
<td>TT</td>
<td>FTTS</td>
<td>FTY</td>
</tr>
<tr>
<td>TS</td>
<td>FTSBM</td>
<td>FTSG</td>
<td>FTSBS</td>
<td>FTSE</td>
<td>FTSH</td>
<td>FTSLB</td>
<td>FTSLM</td>
<td>FTSM</td>
<td>FTSP</td>
<td>FTSS</td>
<td>FTST</td>
<td>TTS</td>
<td>FTSY</td>
</tr>
<tr>
<td>Y</td>
<td>FYBM</td>
<td>FYG</td>
<td>FYBS</td>
<td>FYE</td>
<td>FYH</td>
<td>FYLB</td>
<td>FYLM</td>
<td>FYM</td>
<td>FYP</td>
<td>FYS</td>
<td>FYT</td>
<td>FYTS</td>
<td>TY</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_3_2">
<label>4.3.2</label>
<title>Precision</title>
<p>Precision is the ratio of correctly predicted images to the total predicted images for a class. For example, Precision of black mold class is calculated as per <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref>. It is the ratio of correctly classified samples of black mold (TBM) and the total number of samples classified as black mold, either true or false black mold ones to different categories mentioned in <xref ref-type="table" rid="table-4">Table 4</xref>. Similarly, Precision for each class is calculated individually. The average precision of the model is calculated by combining precision of all individual classes as shown in <xref ref-type="disp-formula" rid="eqn-2">Eq. (2)</xref>.</p>

<p><disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>k</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd /><mml:mtd><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi><mml:mi>M</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>H</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>B</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>P</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>Y</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p><disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mi>A</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>13</mml:mn></mml:mrow></mml:msubsup><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mn>13</mml:mn></mml:mfrac></mml:math></disp-formula></p>
</sec>
<sec id="s4_3_3">
<label>4.3.3</label>
<title>Recall</title>
<p>Recall is the ratio of correctly classified images to the total number of images used for classification. For example, the recall for black mold is calculated as per <xref ref-type="disp-formula" rid="eqn-3">Eq. (3)</xref>. It is the ratio of correctly classified samples of black mold (TBM) and total number of samples of black mold including correctly and incorrectly classified samples. Similarly, Recall is calculated for each disease class individually. Average Recall is calculated by using recall for all the individual classes. The formula is shown in <xref ref-type="disp-formula" rid="eqn-4">Eq. (4)</xref>.</p>
<p><disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd /><mml:mtd><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:msub><mml:mi>l</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>k</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd /><mml:mtd><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi><mml:mi>M</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>G</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>B</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>H</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>L</mml:mi><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>L</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>T</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mi>B</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p><p><disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mrow><mml:mtext>Average Recall&#xA0;</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>13</mml:mn></mml:mrow></mml:msubsup><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:msub><mml:mi>l</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mn>13</mml:mn></mml:mfrac></mml:math></disp-formula></p>
</sec>
<sec id="s4_3_4">
<label>4.3.4</label>
<title>F1 score</title>
<p>F1 score, as shown in <xref ref-type="disp-formula" rid="eqn-5">Eq. (5)</xref>, is calculated by using precision and recall. The F1 score is calculated to evaluate the performance of the model even when it is applied on imbalanced dataset.
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>s</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mrow><mml:mtext>precision</mml:mtext></mml:mrow><mml:mo>&#x00D7;</mml:mo><mml:mrow><mml:mtext>recall</mml:mtext></mml:mrow></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
</sec>
<sec id="s4_3_5">
<label>4.3.5</label>
<title>Accuracy</title>
<p>Accuracy is the total number of correctly classified images from the total number of classified images. Accuracy is calculated in <xref ref-type="disp-formula" rid="eqn-6">Eq. (6)</xref>. It is the ratio of the sum of correctly classified images to black mold (TBM), gray spot (G), bacterial spot (TBS), early blight (E), healthy (H), late blight (TLB), leaf mold (LM), mosaic virus (M), powdery mildew (P), Septoria leaf spot (S), two spotted spider mite (T), target spot (TS), yellow leaf curl virus (Y), and total number of images in the dataset.
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>B</mml:mi><mml:mi>M</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>G</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>B</mml:mi><mml:mi>S</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>E</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>H</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>L</mml:mi><mml:mi>B</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>L</mml:mi><mml:mi>M</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>M</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>S</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>T</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>T</mml:mi><mml:mi>S</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>Y</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>n</mml:mi><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>d</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Results</title>
<p>In this section, we demonstrate the results obtained by applying DL models such as VGG16, ViT, CvT, and Swift-Lite-CvT on the prepared CTL datasets. The confusion matrices obtained for 13 class classification are shown in <xref ref-type="table" rid="table-6">Tables 6</xref>&#x2013;<xref ref-type="table" rid="table-9">9</xref>.</p>
<table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>VGG 16-confusion matrix for 13 class classification</title>
</caption>
<table frame="hsides" >
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th></th>
<th></th>
<th align="center" colspan="13">Predicted class</th>
</tr>
<tr>
<th></th>
<th></th>
<th>BM</th>
<th>G</th>
<th>BS</th>
<th>E</th>
<th>H</th>
<th>LB</th>
<th>LM</th>
<th>M</th>
<th>P</th>
<th>S</th>
<th>T</th>
<th>TS</th>
<th>Y</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="13">Actual class</td>
<td>BM</td>
<td>105</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>G</td>
<td>5</td>
<td>94</td>
<td>1</td>
<td>0</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>3</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>2</td>
</tr>
<tr>
<td>BS</td>
<td>4</td>
<td>4</td>
<td>92</td>
<td>0</td>
<td>1</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>3</td>
<td>0</td>
<td>0</td>
<td>1</td>
</tr>
<tr>
<td>E</td>
<td>0</td>
<td>0</td>
<td>4</td>
<td>98</td>
<td>0</td>
<td>4</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>1</td>
<td>0</td>
</tr>
<tr>
<td>H</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>99</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>7</td>
<td>0</td>
</tr>
<tr>
<td>LB</td>
<td>3</td>
<td>4</td>
<td>4</td>
<td>5</td>
<td>0</td>
<td>83</td>
<td>1</td>
<td>0</td>
<td>6</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>1</td>
</tr>
<tr>
<td>LM</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>106</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>M</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">0</td>
<td>106</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>0</td>
</tr>
<tr>
<td>P</td>
<td>0</td>
<td>9</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td style="background:#FDE9D9;">1</td>
<td>0</td>
<td>97</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>S</td>
<td>0</td>
<td>0</td>
<td>4</td>
<td>2</td>
<td>0</td>
<td>1</td>
<td>2</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td>98</td>
<td>1</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>T</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">1</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td>104</td>
<td>3</td>
<td>0</td>
</tr>
<tr>
<td>TS</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>5</td>
<td>3</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>3</td>
<td>2</td>
<td>93</td>
<td style="background:#FDE9D9;">0</td>
</tr>
<tr>
<td>Y</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">0</td>
<td>106</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-7">
<label>Table 7</label>
<caption>
<title>ViT-confusion matrix for 13 class classification</title>
</caption>
<table frame="hsides" >
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th></th>
<th></th>
<th align="center" colspan="13">Predicted class</th>
</tr>
<tr>
<th></th>
<th></th>
<th>BM</th>
<th>G</th>
<th>BS</th>
<th>E</th>
<th>H</th>
<th>LB</th>
<th>LM</th>
<th>M</th>
<th>P</th>
<th>S</th>
<th>T</th>
<th>TS</th>
<th>Y</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="13">Actual class</td>
<td>BM</td>
<td>108</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>G</td>
<td>0</td>
<td>108</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>BS</td>
<td>1</td>
<td>2</td>
<td>101</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>1</td>
<td>2</td>
</tr>
<tr>
<td>E</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>105</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>3</td>
<td>0</td>
</tr>
<tr>
<td>H</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>108</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>LB</td>
<td>2</td>
<td>5</td>
<td>0</td>
<td>8</td>
<td>2</td>
<td>89</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
</tr>
<tr>
<td>LM</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>105</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>1</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>M</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">0</td>
<td>107</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
</tr>
<tr>
<td>P</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td style="background:#FDE9D9;">0</td>
<td>0</td>
<td>105</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>S</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td>108</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>T</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">2</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td>103</td>
<td>3</td>
<td>0</td>
</tr>
<tr>
<td>TS</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>3</td>
<td>1</td>
<td>102</td>
<td style="background:#FDE9D9;">0</td>
</tr>
<tr>
<td>Y</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">0</td>
<td>108</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-8">
<label>Table 8</label>
<caption>
<title>CvT-confusion matrix for 13 class classification</title>
</caption>
<table frame="hsides" >
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>    
<thead valign="top">
<tr>
<th/>
<th/>
<th align="center" colspan="13">Predicted class</th>
</tr>
<tr>
<th></th>
<th></th>
<th>BM</th>
<th>G</th>
<th>BS</th>
<th>E</th>
<th>H</th>
<th>LB</th>
<th>LM</th>
<th>M</th>
<th>P</th>
<th>S</th>
<th>T</th>
<th>TS</th>
<th>Y</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="13">Actual class</td>
<td>BM</td>
<td>106</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>G</td>
<td>6</td>
<td>94</td>
<td>3</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td>3</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>BS</td>
<td>3</td>
<td>1</td>
<td>99</td>
<td>1</td>
<td>0</td>
<td>1</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>E</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>96</td>
<td>1</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>3</td>
<td>4</td>
<td>0</td>
</tr>
<tr>
<td>H</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>105</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>3</td>
<td>0</td>
</tr>
<tr>
<td>LB</td>
<td>5</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>96</td>
<td>4</td>
<td>1</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>LM</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>2</td>
<td>0</td>
<td>1</td>
<td>99</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>3</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>M</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">0</td>
<td>106</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
</tr>
<tr>
<td>P</td>
<td>2</td>
<td>4</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td style="background:#FDE9D9;">0</td>
<td>0</td>
<td>100</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>S</td>
<td>0</td>
<td>0</td>
<td>4</td>
<td>2</td>
<td>0</td>
<td>2</td>
<td>5</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td>92</td>
<td>0</td>
<td>3</td>
<td>0</td>
</tr>
<tr>
<td>T</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">2</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td>100</td>
<td>3</td>
<td>0</td>
</tr>
<tr>
<td>TS</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>3</td>
<td>4</td>
<td>100</td>
<td style="background:#FDE9D9;">1</td>
</tr>
<tr>
<td>Y</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">0</td>
<td>104</td>
</tr>
</tbody>
</table>
</table-wrap><table-wrap id="table-9">
<label>Table 9</label>
<caption>
<title>Swift-Lite-CvT-confusion matrix for 13 class classification</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th/>
<th/>
<th align="center" colspan="13">Predicted class</th>
</tr>
<tr>
<th></th>
<th></th>
<th>BM</th>
<th>G</th>
<th>BS</th>
<th>E</th>
<th>H</th>
<th>LB</th>
<th>LM</th>
<th>M</th>
<th>P</th>
<th>S</th>
<th>T</th>
<th>TS</th>
<th>Y</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="13">Actual class</td>
<td>BM</td>
<td>105</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>G</td>
<td>0</td>
<td>106</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>BS</td>
<td>0</td>
<td>0</td>
<td>99</td>
<td>2</td>
<td>0</td>
<td>5</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>E</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>97</td>
<td>0</td>
<td>6</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>3</td>
<td>0</td>
</tr>
<tr>
<td>H</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>106</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>LB</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>94</td>
<td>2</td>
<td>3</td>
<td>4</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>3</td>
</tr>
<tr>
<td>LM</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>1</td>
<td>105</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>M</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">0</td>
<td>104</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td>0</td>
<td>0</td>
</tr>
<tr>
<td>P</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>4</td>
<td>0</td>
<td>2</td>
<td style="background:#FDE9D9;">0</td>
<td>0</td>
<td>99</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>1</td>
</tr>
<tr>
<td>S</td>
<td>0</td>
<td>1</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td>104</td>
<td>1</td>
<td>1</td>
<td>0</td>
</tr>
<tr>
<td>T</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>2</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td style="background:#FDE9D9;">0</td>
<td>100</td>
<td>5</td>
<td>0</td>
</tr>
<tr>
<td>TS</td>
<td>0</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>1</td>
<td>2</td>
<td>1</td>
<td style="background:#FDE9D9;">4</td>
<td style="background:#FDE9D9;">0</td>
<td>1</td>
<td>0</td>
<td>93</td>
<td style="background:#FDE9D9;">5</td>
</tr>
<tr>
<td>Y</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>1</td>
<td>0</td>
<td>1</td>
<td>1</td>
<td>0</td>
<td>0</td>
<td>1</td>
<td>2</td>
<td style="background:#FDE9D9;">0</td>
<td>101</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s5_1">
<label>5.1</label>
<title>VGG16</title>
<p>From the confusion matrix shown in <xref ref-type="table" rid="table-6">Table 6</xref>, we observe that gray spot, powdery mildew, late blight, and healthy are primarily misclassified as black mold, gray spot, powdery mildew, and target spot, respectively. The model does the maximum number of misclassifications for late blight disease. This is due to similarities in its visual patterns or with other classes such as early blight and bacterial spot.</p>

</sec>
<sec id="s5_2">
<label>5.2</label>
<title>ViT</title>
<p>From the confusion matrix shown in <xref ref-type="table" rid="table-7">Table 7</xref>, we observe misclassifications of late blight to gray spot and early blight to target spot. The model performs the highest number of misclassifications for late blight disease class.</p>

</sec>
<sec id="s5_3">
<label>5.3</label>
<title>CvT</title>
<p>From the confusion matrix presented in <xref ref-type="table" rid="table-8">Table 8</xref>, we observe that CvT model misclassifies gray spot to black mold, early blight to target spot, late blight to black mold, and septoria leaf spot to leaf mold. Among all the classes, the maximum number of misclassifications are done for gray spot and septoria leaf spot. This is due to similarity in visual symptoms of diseases and a smaller number of training samples of gray spot, early blight, late blight, and Septoria spot.</p>

</sec>
<sec id="s5_4">
<label>5.4</label>
<title>Swift-Lite-CvT</title>
<p>It is evident from the confusion matrices are shown in <xref ref-type="table" rid="table-6">Tables 6</xref>&#x2013;<xref ref-type="table" rid="table-9">9</xref> that VGG16, ViT, CvT, as well as Swift-Lite-CvT, that they perform the maximum number of misclassifications of powdery mildew, bacterial spot, septoria leaf spot and two-spotted spider mite diseases to the early blight, late blight, and target spot disease classes. This is due to the similarity in their symptoms. The sample misclassified images of the early blight, late blight and target spot are shown in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>. Also, the confusion matrix shown in <xref ref-type="table" rid="table-7">Table 7</xref> proves the supremacy of the ViT model with fewer misclassifications.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Sample of misclassified images of tomato plant leaves</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48119-fig-8.tif"/>
</fig>
</sec>
<sec id="s5_5">
<label>5.5</label>
<title>Model Performance</title>
<p>The performance of the implemented models is evaluated for each class individually and in average for all classes.</p>
<sec id="s5_5_1">
<label>5.5.1</label>
<title>Classwise Performance of Swift-Lite-CvT for 13 Class Classification</title>
<p>Using the confusion matrix shown in <xref ref-type="table" rid="table-9">Table 9</xref>, the precision, recall, and F1 score for the &#x201C;Swift-Lite-CvT&#x201D; model were calculated for 13-class classification. The values obtained are presented in <xref ref-type="table" rid="table-10">Table 10</xref>. It is evident from the <xref ref-type="table" rid="table-10">Table 10</xref> that values for precision and recall are more than 90% for all the classes except early blight, late blight, and target spot. Also, the F1 score varies from 93% to 98% except in the above-mentioned three classes. This is due to the similarity in disease symptoms and the highest number of misclassifications. In contrast, the highest values of precision, recall, and F1 score are reported for Healthy, and Gray Spot classes. The performance of the remaining classes is comparable to each other.</p>
<table-wrap id="table-10">
<label>Table 10</label>
<caption>
<title>Swift-Lite-CvT classification report for 13-class classification for each class</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th>Disease name</th>
<th>Precision</th>
<th>Recall</th>
<th>F1 score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td>BM</td>
<td>0.9722</td>
<td>1.0000</td>
<td>0.9859</td>
</tr>
<tr>
<td>G</td>
<td>0.9814</td>
<td>0.9814</td>
<td>0.9814</td>
</tr>
<tr>
<td>BS</td>
<td>0.9166</td>
<td>0.9519</td>
<td>0.9339</td>
</tr>
<tr>
<td>E</td>
<td>0.8981</td>
<td>0.8899</td>
<td>0.8939</td>
</tr>
<tr>
<td>H</td>
<td>0.9814</td>
<td>0.9906</td>
<td>0.9859</td>
</tr>
<tr>
<td>LB</td>
<td>0.8703</td>
<td>0.8173</td>
<td>0.8429</td>
</tr>
<tr>
<td>LM</td>
<td>0.9722</td>
<td>0.9375</td>
<td>0.9545</td>
</tr>
<tr>
<td>M</td>
<td>0.9629</td>
<td>0.9203</td>
<td>0.9411</td>
</tr>
<tr>
<td>P</td>
<td>0.9166</td>
<td>0.9611</td>
<td>0.9611</td>
</tr>
<tr>
<td>S</td>
<td>0.9629</td>
<td>0.9629</td>
<td>0.9629</td>
</tr>
<tr>
<td>T</td>
<td>0.9259</td>
<td>0.9433</td>
<td>0.9345</td>
</tr>
<tr>
<td>TS</td>
<td>0.8611</td>
<td>0.8942</td>
<td>0.8773</td>
</tr>
<tr>
<td>Y</td>
<td>0.9351</td>
<td>0.9181</td>
<td>0.9265</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5_5_2">
<label>5.5.2</label>
<title>Average Performance of DL Models</title>
<p>In this section, we illustrate the average precision, recall, accuracy, and F1 score of VGG16, ViT, CvT, and Swift-Lite-CvT models applied for 10, 11, and 13 class classifications. We also showcase the training time and storage space required by these models. The comparison in average precision, recall, F1 score, and accuracy for all the above-stated models is shown in <xref ref-type="table" rid="table-11">Table 11</xref>.</p>
<table-wrap id="table-11">
<label>Table 11</label>
<caption>
<title>Performance evaluation of deep learning and transformer models applied for classification</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th>Classification</th>
<th>Models</th>
<th align="center" colspan="6">Overall</th>
</tr>
<tr>
<td></td>
<td></td>
<td>Precision</td>
<td>Recall</td>
<td>F1 score</td>
<td>Accuracy</td>
<td>Time (in min)</td>
<td>Size (MB)</td>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4">10 class classification</td>
<td>VGG16</td>
<td>1.00</td>
<td>0.999</td>
<td>1.00</td>
<td>99.87%</td>
<td>1367</td>
<td>537.3</td>
</tr>
<tr>
<td>ViT</td>
<td>1.00</td>
<td>1.00</td>
<td>1.00</td>
<td>99.94%</td>
<td>2459</td>
<td>345.7</td>
</tr>
<tr>
<td>CvT</td>
<td>1.00</td>
<td>1.00</td>
<td>1.00</td>
<td>99.81%</td>
<td>1450</td>
<td>78.8</td>
</tr>
<tr>
<td>Swift-Lite-CvT</td>
<td>0.996</td>
<td>0.996</td>
<td>0.994</td>
<td><bold>99.45%</bold></td>
<td><bold>458</bold></td>
<td><bold>5.8</bold></td>
</tr>
<tr>
<td rowspan="4">11 class classification</td>
<td>VGG16</td>
<td>0.9790</td>
<td>0.9809</td>
<td>0.9790</td>
<td>97.81%</td>
<td>198</td>
<td>537.3</td>
</tr>
<tr>
<td>ViT</td>
<td>0.9718</td>
<td>0.9718</td>
<td>0.9718</td>
<td>99.37%</td>
<td>338</td>
<td>345.7</td>
</tr>
<tr>
<td>CvT</td>
<td>0.8863</td>
<td>0.8872</td>
<td>0.8863</td>
<td>95.72%</td>
<td>208</td>
<td>78.8</td>
</tr>
<tr>
<td>Swift-Lite-CvT</td>
<td>0.9636</td>
<td>0.9645</td>
<td>0.9663</td>
<td><bold>96.45%</bold></td>
<td><bold>96</bold></td>
<td><bold>5.8</bold></td>
</tr>
<tr>
<td rowspan="4">13 class classification</td>
<td>VGG16</td>
<td>0.9123</td>
<td>0.9061</td>
<td>0.9091</td>
<td>90.12%</td>
<td>90</td>
<td>537.3</td>
</tr>
<tr>
<td>ViT</td>
<td>0.9653</td>
<td>0.9669</td>
<td>0.9661</td>
<td>96.66%</td>
<td>170</td>
<td>345.7</td>
</tr>
<tr>
<td>CvT</td>
<td>0.9237</td>
<td>0.9256</td>
<td>0.9246</td>
<td>92.37%</td>
<td>104</td>
<td>78.8</td>
</tr>
<tr>
<td>Swift-Lite-CvT</td>
<td>0.9351</td>
<td>0.9437</td>
<td>0.9393</td>
<td><bold>93.51%</bold></td>
<td><bold>44</bold></td>
<td><bold>5.8</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>It is evident from <xref ref-type="table" rid="table-11">Table 11</xref> that the Swift-Lite-CvT model achieved an accuracy of 93.51%. It surpasses the 90.4% accuracy of the basic CvT architecture. It is also apparent from <xref ref-type="table" rid="table-11">Table 11</xref> that the accuracy of all the models viz. VGG 16, ViT, CvT, and Swift-Lite-CvT decreases with an increase in a number of classes. The decrease is due to the availability of a smaller number of training samples for classes such as gray spot, black mold, and powdery mildew. The precision of the Swift-Lite-CvT model decreases with an increase in a number of classes. This is due to the increase in data imbalance. In 13 class datasets, there are 67, 53, and 125 images for gray spot, black mold, and powdery mildew, respectively. This adversely affects the training of the model. However, we increased the number of samples by data augmentation, the number of training samples is 3900, 1004, and 428 for 10 class, 11 class, and 13 class, respectively. The variation in a number of training samples leads to variation in the precision of the model. Additionally, we observed that the proposed model requires storage of 5.8 MB. The storage requirement ratio of the Swift-Lite-CvT is 1:98 with VGG16, 2:98 with ViT, and 7:92 with CvT. Moreover, the proposed models utilize a training time of 44 min. The training time ratio for the proposed model is 48:51 with VGG16, 26:74 with ViT and 42:57 with CvT model. Thus, it is apparent that the proposed model is time and storage-efficient than state-of-the-art models.</p>

<p>Furthermore, our analysis revealed that increasing the number of classes for training affects the model&#x2019;s performance. A degradation in accuracy is depicted as the increase in number of classes from 10 to 13 as shown in <xref ref-type="table" rid="table-11">Table 11</xref>. Specifically, when conducting a 10-class classification task, we observed remarkably high accuracy across all implemented models. However, as we transitioned to 11-class and 13-class classification tasks, we observed a decline in accuracy. In order to address this challenge, we proposed the utilization of the transformer architecture for the 13-class classification task. The implementation of this architecture significantly improved results, with an accuracy of 93.51%. Notably, this approach offers the added benefits of reduced storage space requirements and decreased processing time compared to previous models. It is evident from results reported in <xref ref-type="fig" rid="fig-9">Fig. 9</xref> that a reduction of 51.12%, 74.12%, and 57.7% in computing time of Swift-Lite-CvT is reported when compared to VGG16, ViT, and CvT models, respectively. Moreover, a reduction of 98.93% 98.33%, and a 92.64% in storage space of Swift-Lite-CvT is reported when compared to VGG16, ViT, and CvT models.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>Comparison in storage space and computation time required by the implemented deep learning and transformer models</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_48119-fig-9.tif"/>
</fig>
</sec>
</sec>
<sec id="s5_6">
<label>5.6</label>
<title>Comparative Analysis</title>
<p>To further validate the effectiveness of the Swift-Lite-CvT model, we conducted a comprehensive performance comparison with models applied in the literature for the detection of tomato diseases [<xref ref-type="bibr" rid="ref-51">51</xref>&#x2013;<xref ref-type="bibr" rid="ref-53">53</xref>]. For instance, the models proposed in references [<xref ref-type="bibr" rid="ref-32">32</xref>&#x2013;<xref ref-type="bibr" rid="ref-34">34</xref>] achieved comparable accuracies of 96.30%, 99.64%, and 98.05%, respectively. These models utilized a combination of CNN and transformer architectures for 10 class classification. Similarly, Hossain et al., as outlined in reference [<xref ref-type="bibr" rid="ref-54">54</xref>], applied a ViT transformer model with 11 classes and attained an accuracy of 97%, aligning closely with the performance of our proposed model. However, it is noteworthy that the accuracy experienced a decline of 6% as the number of classes increased to 13. The comparison is illustrated in <xref ref-type="table" rid="table-12">Table 12</xref>.</p>
<table-wrap id="table-12">
<label>Table 12</label>
<caption>
<title>Comparison of results of Swift-Lite-CvT model against other deep learning and transformer models</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead valign="top">
<tr>
<th>Authors</th>
<th>Year</th>
<th>Methods</th>
<th>Classification</th>
<th>Accuracy</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td>Agarwal et al. [<xref ref-type="bibr" rid="ref-23">23</xref>]</td>
<td>2020</td>
<td>Proposed CNN model</td>
<td>10 classes</td>
<td>91.20%</td>
</tr>
<tr>
<td>Ahmad et al. [<xref ref-type="bibr" rid="ref-24">24</xref>]</td>
<td>2020</td>
<td>CNN model inception V3</td>
<td>6 classes</td>
<td>93.70%</td>
</tr>
<tr>
<td>Zhou et al. [<xref ref-type="bibr" rid="ref-30">30</xref>]</td>
<td>2021</td>
<td>Restructured deep residual dense network</td>
<td>9 classes</td>
<td>95%</td>
</tr>
<tr>
<td>Chowdhary et al. [<xref ref-type="bibr" rid="ref-26">26</xref>]</td>
<td>2021</td>
<td>DenseNet201 DL model</td>
<td>2 classes, 6 classes, and 10 classes</td>
<td>2 Class&#x2013;99.2%, 6 Class&#x2013;97.99%, 10 Class&#x2013;98.05%</td>
</tr>
<tr>
<td>Wang et al. [<xref ref-type="bibr" rid="ref-34">34</xref>]</td>
<td>2022</td>
<td>ViT &#x002B; CNN</td>
<td>10 classes</td>
<td>96.30%</td>
</tr>
<tr>
<td>Alzahrani et al. [<xref ref-type="bibr" rid="ref-40">40</xref>]</td>
<td>2023</td>
<td>Proposed DenseNet121 model &#x002B; ViT</td>
<td>10 classes</td>
<td>99.64%</td>
</tr>
<tr>
<td>Hossain et al. [<xref ref-type="bibr" rid="ref-54">54</xref>]</td>
<td>2023</td>
<td>Max ViT</td>
<td>11 classes</td>
<td>97%</td>
</tr>
<tr>
<td colspan="2"><bold>Proposed Swift-Lite-CvT model</bold></td>
<td><bold>Swift-Lite-CvT</bold></td>
<td><bold>10 classes, 11 classes and 13 classes</bold></td>
<td><bold>10 class&#x2013;99.45%, 11 class&#x2013;96.45%, 13 class&#x2013;93.51%</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In our investigation, we found out that the Swift-Lite-CvT model has reached its highest accuracy of 99.45% in the case of 10 class classification. Also, it maintains a commendable accuracy of 96.45% in the 11-class classification and a competitive 93.51% in the 13-class classification. While there is room for improvement in terms of accuracy, it is important to highlight that our model demonstrates a significant reduction in both computation time and storage requirements. This improved efficiency enhances the acceptability of our model in real-life implementation for tomato disease classification.</p>
</sec>
</sec>
<sec id="s6">
<label>6</label>
<title>Discussions</title>
<p>The CvT model offers a unique blend of convolutional and transformer elements. Convolutions excel at capturing local spatial details, whereas transformers empower the model to analyze global context. This dual proficiency renders CvT a resilient model for visual tasks, surpassing models that rely solely on one of these two mechanisms. The alterations made in the Swift-Lite-CvT were aimed at enhancingcomputational efficiency while preserving a substantial portion of the model&#x2019;s representational capacity. Here, a convolutional block is used instead of the linear patch projection as in ViT model. This allows for the extraction of local features and spatial information in a more efficient way, especially when dealing with smaller models. Additionally, replacing positional encoding with a convolutional block helps alleviate the need for explicit encoding of positional information. Convolutional layers inherently capture spatial relationships and patterns, removing the need for separate positional encoding. By decreasing the depth of the transformer blocks and fine tuning the number of attention heads, the Swift-Lite-CvT achieved a notable reduction in training by up to 58% which is approximately two times faster than the original CvT The study highlights advantages such as efficient resource utilization and reduced training time making it suitable for mobile devices.</p>
<p>The study given in this manuscript highlights the effectiveness of the Swift-Lite-CvT model for classifying tomato plant leaves. It achieves high accuracy while demanding significantly less storage space and computational time compared to the ViT model as shown in <xref ref-type="fig" rid="fig-9">Fig. 9</xref>. The proposed architecture utilizes only 44 min and a mere 5.8 MB of space, marking a remarkable success in the realm of plant leaf disease detection. In this study, modifying the heads and depths in the proposed hybrid transformer architecture makes the model efficient even with a small dataset.</p>

</sec>
<sec id="s7">
<label>7</label>
<title>Conclusions</title>
<p>In this manuscript, we successfully developed the deep learning model &#x201C;Swift-Lite-CvT&#x201D; for accurately classifying diseases using images of tomato leaves. We achieved the objectives of minimizing the storage space, and reducing the training time, and some parameters without compromising the accuracy of the model. The model takes a storage space of 5.8 MB and a training time of 44 min. It proves its applicability in real-life scenarios on mobile devices. Moreover, the ratio of storage space of the proposed model to that of VGG16, ViT, and CvT is 1:98, 2:98, and 7:92, respectively. Also, the training time ratio of the proposed model with VGG16, ViT, and CvT models is 48:51, 26:74, and 42:57, respectively. This justifies the superiority of the proposed model &#x201C;Swift-Lite-CvT&#x201D; over the above-mentioned models. The model is efficient in handling versatile datasets comprising images with plain as well as complex backgrounds. The efficacy of the proposed model for 10, 11, and 13 classes as shown in <xref ref-type="table" rid="table-11">Table 11</xref>, proves the robustness of the model. The comparison in VGG16, ViT, CvT, and Swift-Lite-CvT models for 10, 11, and 13 classes shows the highest accuracy achieved for 10-class classification. The decrease in accuracy is reported when the number of classes increases from 10 to 11, and 13. The decrease is due to data imbalance and availability of merely 67, 53, and 125 training samples in the original dataset for gray spot, black mold, and powdery mildew classes, respectively. Among all the above-mentioned models, ViT reported the highest accuracy of 96.66%, it utilized a substantial amount of storage space 345.7 MB and a training time of 170 minutes. In contrast, the Swift-Lite-CvT model achieved a comparable accuracy of 93.51%, utilized a storage space of 5.8 MB and a training time of 44 minutes. This plain and compact model can be easily integrated with a mobile or web application and can be implemented in real-life scenarios for classifying tomato leaf diseases. It can also be implemented using IoT-enabled devices, integrated with a user interface for real-time monitoring of tomato fields, and disease detection at an early stage.</p>

</sec>
</body>
<back>
<ack>
<p>We acknowledge the lab support given by Manipal University Jaipur in carrying out this research.</p>
</ack>
<sec><title>Funding Statement</title>
<p>This work is supported by the Department of Informatics, Modeling, Electronics and Systems (DIMES), University of Calabria (Grant/Award Number: SIMPATICO_ZUMPANO).</p>
</sec>
<sec><title>Author Contributions</title>
<p>All authors contributed equally to the study, conception, design, experiments, manuscript writing, and preparation of the manuscript. All authors reviewed and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability"><title>Availability of Data and Materials</title>
<p>The data with annotations is available with the authors. We will make the data available after receiving review comments for the manuscript.</p>
</sec>
<sec sec-type="COI-statement"><title>Conflicts of Interest</title>
<p>The authors declare that they have no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>S. C.</given-names> <surname>Fran&#x00E7;ois-Xavier Branth&#x00F4;me</surname></string-name>, and <string-name><given-names>Madeleine</given-names> <surname>Roy&#x00E8;re-Koonings</surname></string-name></person-group>, &#x201C;<article-title>tomato online conference</article-title>,&#x201D; <publisher-loc>Parma, Italy</publisher-loc>, <year>2020</year>. <comment>Accessed: Jan. 17, 2024</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="Https://www.tomatonews.com/en/worldwide-total-fresh-tomato-production-exceeds-187-million-tonnes-in-2020_2_1565.html">https://www.tomatonews.com/en/worldwide-total-fresh-tomato-production-exceeds-187-million-tonnes-in-2020_2_1565.html</ext-link></mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="other"><collab>Worldostats team</collab>, &#x201C;<article-title>Tomato Production by Country 2023</article-title>,&#x201D; <year>2023</year>, <comment>Accessed: Mar. 27, 2024</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.worldostats.com/post/tomato-production-by-country-2023">https://www.worldostats.com/post/tomato-production-by-country-2023</ext-link></mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>More</surname></string-name></person-group>, &#x201C;<article-title>Business standard</article-title>,&#x201D; pp. <fpage>1</fpage>&#x2013;<lpage>7</lpage>, <year>2021</year>. <comment>Accessed: Oct. 18, 2023</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://timesofindia.indiatimes.com/potato-and-tomato-output-projected-to-be-down-by-4-5-pc-in-2021-22-govt-data/articleshow/95124336.cms">https://timesofindia.indiatimes.com/potato-and-tomato-output-projected-to-be-down-by-4-5-pc-in-2021-22-govt-data/articleshow/95124336.cms</ext-link></mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Keelery</surname></string-name></person-group>, &#x201C;<article-title>Production volume of tomatoes</article-title>,&#x201D; <source>Statista</source>, <year>2023</year>. <comment>Accessed: Aug. 24, 2024</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.statista.com/statistics/1039712/india-production-volume-of-tomatoes/">https://www.statista.com/statistics/1039712/india-production-volume-of-tomatoes/</ext-link></mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>van Heghe</surname></string-name> and <string-name><given-names>C.</given-names> <surname>Sirois</surname></string-name></person-group>, &#x201C;<article-title>Tomato Processing Market: Global Industry Trends, Share, Size, Growth, Opportunity and Forecast 2023&#x2013;2028</article-title>,&#x201D; <year>2023</year>, <comment>Accessed: Dec. 23, 2023</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.giiresearch.com/report/imarc1206560-tomato-processing-market-global-industry-trends.html?">https://www.giiresearch.com/report/imarc1206560-tomato-processing-market-global-industry-trends.html?</ext-link></mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><collab>International Market Analysis Research and Consulting Group</collab>, &#x201C;<article-title>Tomato Processing Market: Industry Trends &#x0026; Forecast 2024-2032</article-title>,&#x201D; <year>2023</year>, <comment>Accessed: Dec. 22, 2023</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.imarcgroup.com/tomato-processing-plant">https://www.imarcgroup.com/tomato-processing-plant</ext-link></mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>P. S. C.</given-names> <surname>Shivani Machha</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Jadhav</surname></string-name>, and <string-name><given-names>H.</given-names> <surname>Kasar</surname></string-name></person-group>, &#x201C;<article-title>Crop leaf disease diagnosis using convolutional neural network</article-title>,&#x201D; <source>Int. J. Trend Sci. Res. Dev.</source>, vol. <volume>4</volume>, no. <issue>2</issue>, pp. <fpage>1056</fpage>&#x2013;<lpage>1058</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Kaur</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Varalakshmi</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Pitchaimuthu</surname></string-name>, and <string-name><given-names>B.</given-names> <surname>Mahesha</surname></string-name></person-group>, &#x201C;<article-title>Screening Luffa germplasm and advanced breeding lines for resistance to Tomato leaf curl New Delhi virus</article-title>,&#x201D; <source>J. Gen. Plant Pathol.</source>, vol. <volume>87</volume>, no. <issue>5</issue>, pp. <fpage>287</fpage>&#x2013;<lpage>294</lpage>, <year>Sep. 2021</year>. doi: <pub-id pub-id-type="doi">10.1007/s10327-021-01010-z</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Panno</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>A review of the most common and economically important diseases that undermine the cultivation of tomato crop in the mediterranean basin</article-title>,&#x201D; <source>Agronomy</source>, vol. <volume>11</volume>, no. <issue>11</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>45</lpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.3390/agronomy11112188</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Zhang</surname></string-name>, and <string-name><given-names>B. I. N.</given-names> <surname>Wang</surname></string-name></person-group>, &#x201C;<article-title>Plant disease detection and classification by deep learning&#x2013;A review</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>9</volume>, pp. <fpage>56683</fpage>&#x2013;<lpage>56698</lpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2021.3069646.</pub-id></mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Liu</surname></string-name> and <string-name><given-names>X.</given-names> <surname>Wang</surname></string-name></person-group>, &#x201C;<article-title>Early recognition of tomato gray leaf spot disease based on MobileNetv2-YOLOv3 model</article-title>,&#x201D; <source>Plant Methods</source>, vol. <volume>16</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>16</lpage>, <year>2020</year>. doi: <pub-id pub-id-type="doi">10.1186/s13007-020-00624-2.</pub-id>; <pub-id pub-id-type="pmid">32523613</pub-id></mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>V. S.</given-names> <surname>Dhaka</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>A survey of deep convolutional neural networks applied for prediction of plant leaf diseases</article-title>,&#x201D; <source>Sensors</source>, vol. <volume>21</volume>, no. <issue>14</issue>, pp. <fpage>4749</fpage>, <year>Jul. 2021</year>. doi: <pub-id pub-id-type="doi">10.3390/S21144749.</pub-id>; <pub-id pub-id-type="pmid">34300489</pub-id></mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Mohanty</surname></string-name></person-group>, &#x201C;<article-title>Plant Village Dataset</article-title>,&#x201D; <year>2018</year>, <comment>Accessed: Dec. 22, 2023</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://github.com/spMohanty/PlantVillage-Dataset">https://github.com/spMohanty/PlantVillage-Dataset</ext-link></mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. P.</given-names> <surname>Mohanty</surname></string-name>, <string-name><given-names>D. P.</given-names> <surname>Hughes</surname></string-name>, and <string-name><given-names>M.</given-names> <surname>Salath&#x00E9;</surname></string-name></person-group>, &#x201C;<article-title>Using deep learning for image-based plant disease detection</article-title>,&#x201D; <source>Front. Plant Sci.</source>, vol. <volume>7</volume>, no. <issue>September</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>10</lpage>, <year>2016</year>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2016.01419.</pub-id>; <pub-id pub-id-type="pmid">27713752</pub-id></mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K. P.</given-names> <surname>Ferentinos</surname></string-name></person-group>, &#x201C;<article-title>Deep learning models for plant disease detection and diagnosis</article-title>,&#x201D; <source>Comput. Electron. Agric.</source>, vol. <volume>145</volume>, pp. <fpage>311</fpage>&#x2013;<lpage>318</lpage>, <year>2018</year>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2018.01.009.</pub-id></mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J. G. A.</given-names> <surname>Barbedo</surname></string-name></person-group>, &#x201C;<article-title>Factors influencing the use of deep learning for plant disease recognition</article-title>,&#x201D; <source>Biosyst. Eng.</source>, vol. <volume>172</volume>, no. <issue>660</issue>, pp. <fpage>84</fpage>&#x2013;<lpage>91</lpage>, <year>2018</year>. doi: <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2018.05.013.</pub-id></mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Yuan</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>Tokens-to-token ViT: Training vision transformers from scratch on ImageNet</article-title>,&#x201D; in <source>Proc IEEE Int. Conf. Comput. Vis.</source>, vol. <volume>30</volume>, pp. <fpage>538</fpage>&#x2013;<lpage>547</lpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00060.</pub-id></mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Wu</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>CvT: Introducing convolutions to vision transformers</article-title>,&#x201D; in <source>Proc. IEEE Int. Conf. Comput. Vis.</source>, vol. <volume>6</volume>, pp. <fpage>22</fpage>&#x2013;<lpage>31</lpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00009.</pub-id></mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="other"><collab>Daphne Cornelisse</collab>, &#x201C;<article-title>An intuitive guide to Convolutional Neural Network</article-title>,&#x201D; <year>2018</year>, <comment>Accessed: Dec. 22, 2023</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.freecodecamp.org/news/an-intuitive-guide-to-convolutional-neural-networks-260c2de0a050/">https://www.freecodecamp.org/news/an-intuitive-guide-to-convolutional-neural-networks-260c2de0a050/</ext-link></mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J. G.</given-names> <surname>Arnal Barbedo</surname></string-name></person-group>, &#x201C;<article-title>Plant disease identification from individual lesions and spots using deep learning</article-title>,&#x201D; <source>Biosyst. Eng.</source>, vol. <volume>180</volume>, pp. <fpage>96</fpage>&#x2013;<lpage>107</lpage>, <year>2016</year>. doi: <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2019.02.002.</pub-id></mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>P.</given-names> <surname>Kaur</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>Recognition of leaf disease using hybrid convolutional neural network by applying feature reduction</article-title>,&#x201D; <source>Sensors 2022</source>, vol. <volume>22</volume>, no. <issue>2</issue>, pp. <fpage>575</fpage>, <year>Jan. 2022</year>. doi: <pub-id pub-id-type="doi">10.3390/S22020575.</pub-id>; <pub-id pub-id-type="pmid">35062534</pub-id></mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Thangaraj</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Anandamurugan</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Pandiyan</surname></string-name>, and <string-name><given-names>V. K.</given-names> <surname>Kaliappan</surname></string-name></person-group>, &#x201C;<article-title>Artificial intelligence in tomato leaf disease detection: A comprehensive review and discussion</article-title>,&#x201D; <source>J. Plant. Dis. Prot.</source>, vol. <volume>129</volume>, no. <issue>3</issue>, pp. <fpage>469</fpage>&#x2013;<lpage>488</lpage>, <year>2022</year>. doi: <pub-id pub-id-type="doi">10.1007/s41348-021-00500-8.</pub-id></mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Agarwal</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Singh</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Arjaria</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Sinha</surname></string-name>, and <string-name><given-names>S.</given-names> <surname>Gupta</surname></string-name></person-group>, &#x201C;<article-title>ToLeD: Tomato leaf disease detection using convolution neural network</article-title>,&#x201D; <source>Procedia Comput. Sci.</source>, vol. <volume>167</volume>, pp. <fpage>293</fpage>&#x2013;<lpage>301</lpage>, <year>2020</year>. doi: <pub-id pub-id-type="doi">10.1016/j.procs.2020.03.225.</pub-id></mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>I.</given-names> <surname>Ahmad</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Hamid</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Yousaf</surname></string-name>, <string-name><given-names>S. T.</given-names> <surname>Shah</surname></string-name>, and <string-name><given-names>M. O.</given-names> <surname>Ahmad</surname></string-name></person-group>, &#x201C;<article-title>Optimizing pretrained convolutional neural networks for tomato leaf disease detection</article-title>,&#x201D; <source>Complexity</source>, vol. <volume>2020</volume>, pp. <fpage>1</fpage>&#x2013;<lpage>6</lpage>, <year>2020</year>. doi: <pub-id pub-id-type="doi">10.1155/2020/8812019.</pub-id></mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Norria</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ricky</surname></string-name>, and <string-name><given-names>A.</given-names> <surname>Nazari</surname></string-name></person-group>, &#x201C;<article-title>Tomato leaf disease detection using convolution neural network (CNN)</article-title>,&#x201D; <source>Evol. Electr. Electron. Eng.</source>, vol. <volume>2</volume>, no. <issue>2</issue>, pp. <fpage>667</fpage>&#x2013;<lpage>676</lpage>, <year>Nov. 2021</year>. doi: <pub-id pub-id-type="doi">10.30880/eeee.2021.02.02.080.</pub-id></mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. E. H.</given-names> <surname>Chowdhury</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>Automatic and reliable leaf disease detection using deep learning techniques</article-title>,&#x201D; <source>AgriEngineering</source>, vol. <volume>3</volume>, no. <issue>2</issue>, pp. <fpage>294</fpage>&#x2013;<lpage>312</lpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.3390/agriengineering3020020.</pub-id></mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>V.</given-names> <surname>Gonzalez-Huitron</surname></string-name>, <string-name><given-names>J. A.</given-names> <surname>Le&#x00F3;n-Borges</surname></string-name>, <string-name><given-names>A. E.</given-names> <surname>Rodriguez-Mata</surname></string-name>, <string-name><given-names>L. E.</given-names> <surname>Amabilis-Sosa</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Ram&#x00ED;rez-Pereda</surname></string-name> and <string-name><given-names>H.</given-names> <surname>Rodriguez</surname></string-name></person-group>, &#x201C;<article-title>Disease detection in tomato leaves via CNN with lightweight architectures implemented in Raspberry Pi 4</article-title>,&#x201D; <source>Comput. Electron. Agric.</source>, vol. <volume>181</volume>, pp. <fpage>105951</fpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2020.105951.</pub-id></mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. M.</given-names> <surname>Hassan</surname></string-name>, <string-name><given-names>A. K.</given-names> <surname>Maji</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Jasi&#x0144;ski</surname></string-name>, <string-name><given-names>Z.</given-names> <surname>Leonowicz</surname></string-name>, and <string-name><given-names>E.</given-names> <surname>Jasi&#x0144;ska</surname></string-name></person-group>, &#x201C;<article-title>Identification of plant-leaf diseases using cnn and transfer-learning approach</article-title>,&#x201D; <source>Electron.</source>, vol. <volume>10</volume>, no. <issue>12</issue>, pp. <fpage>1388</fpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.3390/electronics10121388.</pub-id></mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Abbas</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Jain</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Gour</surname></string-name>, and <string-name><given-names>S.</given-names> <surname>Vankudothu</surname></string-name></person-group>, &#x201C;<article-title>Tomato plant disease detection using transfer learning with C-GAN synthetic images</article-title>,&#x201D; <source>Comput. Electron. Agric.</source>, vol. <volume>187</volume>, pp. <fpage>106279</fpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2021.106279.</pub-id></mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Zhou</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Zhou</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Xing</surname></string-name>, and <string-name><given-names>J.</given-names> <surname>Song</surname></string-name></person-group>, &#x201C;<article-title>Tomato leaf disease identification by restructured deep residual dense network</article-title>,&#x201D; <source>IEEE Access</source>, vol. <volume>9</volume>, pp. <fpage>28822</fpage>&#x2013;<lpage>28831</lpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.1109/ACCESS.2021.3058947.</pub-id></mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A. S.</given-names> <surname>Paymode</surname></string-name> and <string-name><given-names>V. B.</given-names> <surname>Malode</surname></string-name></person-group>, &#x201C;<article-title>Transfer learning for multi-crop leaf disease image classification using convolutional neural network VGG</article-title>,&#x201D; <source>Artif. Intell. Agric.</source>, vol. <volume>6</volume>, no. <issue>2</issue>, pp. <fpage>23</fpage>&#x2013;<lpage>33</lpage>, <year>2022</year>. doi: <pub-id pub-id-type="doi">10.1016/j.aiia.2021.12.002.</pub-id></mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Vadivel</surname></string-name> and <string-name><given-names>R.</given-names> <surname>Suguna</surname></string-name></person-group>, &#x201C;<article-title>Automatic recognition of tomato leaf disease using fast enhanced learning with image processing</article-title>,&#x201D; <source>Acta Agric. Scand. Sect. B Soil Plant Sci.</source>, vol. <volume>72</volume>, no. <issue>1</issue>, pp. <fpage>312</fpage>&#x2013;<lpage>324</lpage>, <year>2022</year>. doi: <pub-id pub-id-type="doi">10.1080/09064710.2021.1976266.</pub-id></mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Tarek</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Aly</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Eisa</surname></string-name>, and <string-name><given-names>M.</given-names> <surname>Abul-Soud</surname></string-name></person-group>, &#x201C;<article-title>Optimized deep learning algorithms for tomato leaf disease detection with hardware deployment</article-title>,&#x201D; <source>Electron.</source>, vol. <volume>11</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>19</lpage>, <year>2022</year>. doi: <pub-id pub-id-type="doi">10.3390/electronics11010140.</pub-id></mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Chen</surname></string-name>, and <string-name><given-names>D.</given-names> <surname>Wang</surname></string-name></person-group>, &#x201C;<article-title>Convolution network enlightened transformer for regional crop disease classification</article-title>,&#x201D; <source>Electron.</source>, vol. <volume>11</volume>, no. <issue>19</issue>, pp. <fpage>3174</fpage>, <year>Oct. 2022</year>. doi: <pub-id pub-id-type="doi">10.3390/ELECTRONICS11193174.</pub-id></mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Yu</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Xie</surname></string-name>, and <string-name><given-names>Q.</given-names> <surname>Huang</surname></string-name></person-group>, &#x201C;<article-title>Inception convolutional vision transformers for plant disease identification</article-title>,&#x201D; <source>Internet of Things</source>, vol. <volume>21</volume>, no. <issue>288</issue>, pp. <fpage>100650</fpage>, <year>2023</year>. doi: <pub-id pub-id-type="doi">10.1016/j.iot.2022.100650.</pub-id></mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>D. P.</given-names> <surname>Hughes</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Salathe</surname></string-name></person-group>, &#x201C;<article-title>An open access repository of images on plant health to enable the development of mobile disease diagnostics</article-title>,&#x201D; arXiv:1511.08060, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><collab>MIT</collab></person-group>, &#x201C;<article-title>iBean</article-title>,&#x201D; <source>Makerere AI Lab</source>, <year>2020</year>, <comment>Accessed: Jan. 20, 2020</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://github.com/AILab-Makerere/ibean/blob/master/README.md">https://github.com/AILab-Makerere/ibean/blob/master/README.md</ext-link></mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Huang</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Liu</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Qi</surname></string-name>, and <string-name><given-names>K.</given-names> <surname>Yang</surname></string-name></person-group>, &#x201C;<article-title>Development and validation of a deep learning algorithm for the recognition of plant disease</article-title>,&#x201D; in <conf-name>Proc. 21st IEEE Int. Conf. High Perform. Comput. Commun.; 17th IEEE Int. Conf. Smart City; 5th IEEE Int. Conf. Data Sci. Syst. (HPCC/SmartCity/DSS)</conf-name>, <year>2019</year>, pp. <fpage>1951</fpage>&#x2013;<lpage>1957</lpage>. doi: <pub-id pub-id-type="doi">10.1109/HPCC/SmartCity/DSS.2019.00269.</pub-id></mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Singh</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Jain</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Jain</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Kayal</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Kumawat</surname></string-name> and <string-name><given-names>N.</given-names> <surname>Batra</surname></string-name></person-group>, &#x201C;<article-title>PlantDoc: A dataset for visual plant disease detection</article-title>,&#x201D; in <source>ACM Int. Conf. Proceeding Ser.</source>, <year>2020</year>, pp. <fpage>249</fpage>&#x2013;<lpage>253</lpage>. doi: <pub-id pub-id-type="doi">10.1145/3371158.</pub-id></mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. S.</given-names> <surname>Alzahrani</surname></string-name> and <string-name><given-names>F. W.</given-names> <surname>Alsaade</surname></string-name></person-group>, &#x201C;<article-title>Transform and deep learning algorithms for the early detection and recognition of tomato leaf disease</article-title>,&#x201D; <source>Agronomy</source>, vol. <volume>13</volume>, no. <issue>5</issue>, pp. <fpage>1184</fpage>, <year>2023</year>. doi: <pub-id pub-id-type="doi">10.3390/agronomy13051184.</pub-id></mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Bao</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Huang</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Hu</surname></string-name>, and <string-name><given-names>D.</given-names> <surname>Liang</surname></string-name></person-group>, &#x201C;<article-title>Identification of maize leaf diseases using improved convolutional neural network</article-title>,&#x201D; <source>Nongye Gongcheng Xuebao/Transactions Chinese Soc. Agric. Eng.</source>, vol. <volume>37</volume>, no. <issue>6</issue>, pp. <fpage>160</fpage>&#x2013;<lpage>167</lpage>, <year>2021</year>. doi: <pub-id pub-id-type="doi">10.11975/j.issn.1002-6819.2021.06.020.</pub-id></mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Li</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Zhang</surname></string-name>, and <string-name><given-names>B.</given-names> <surname>Wang</surname></string-name></person-group>, &#x201C;<article-title>Apple leaf disease identification with a small and imbalanced dataset based on lightweight convolutional networks</article-title>,&#x201D; <source>Sens.</source>, vol. <volume>22</volume>, no. <issue>1</issue>, pp. <fpage>173</fpage>, <year>2022</year>. doi: <pub-id pub-id-type="doi">10.3390/s22010173.</pub-id>; <pub-id pub-id-type="pmid">35009716</pub-id></mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Abdallah Ali</surname></string-name></person-group>, &#x201C;<article-title>PlantVillage Dataset</article-title>,&#x201D; <year>2019</year>, <comment>Accessed: Dec. 25, 2023</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/abdallahalidev/plantvillage-dataset">https://www.kaggle.com/datasets/abdallahalidev/plantvillage-dataset</ext-link></mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y. H.</given-names> <surname>Huang</surname></string-name> and <string-name><given-names>M. L.</given-names> <surname>Chang</surname></string-name></person-group>, &#x201C;<article-title>Dataset of tomato leaves</article-title>,&#x201D; <source>Mendeley Data</source>, <year>2020</year>. doi: <pub-id pub-id-type="doi">10.17632/ngdgg79rzb.1.</pub-id></mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Nirmal Sankalana</surname></string-name></person-group>, &#x201C;<article-title>PlantDoc Classification dataset</article-title>,&#x201D; <year>2020</year>, <comment>Accessed: Dec. 25, 2023</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/nirmalsankalana/plantdoc-dataset">https://www.kaggle.com/datasets/nirmalsankalana/plantdoc-dataset</ext-link></mixed-citation></ref>
<ref id="ref-46"><label>[46]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Simonyan</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Zisserman</surname></string-name></person-group>, &#x201C;<article-title>Very deep convolutional networks for large-scale image recognition</article-title>,&#x201D; in <conf-name>3rd Int. Conf. Learn. Represent.</conf-name>, <year>2015</year>, pp. <fpage>1</fpage>&#x2013;<lpage>14</lpage>.</mixed-citation></ref>
<ref id="ref-47"><label>[47]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. H.</given-names> <surname>Lee</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Go&#x00EB;au</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Bonnet</surname></string-name>, and <string-name><given-names>A.</given-names> <surname>Joly</surname></string-name></person-group>, &#x201C;<article-title>New perspectives on plant disease characterization based on deep learning</article-title>,&#x201D; <source>Comput. Electron. Agric.</source>, vol. <volume>170</volume>, pp. <fpage>105220</fpage>, <year>2020</year>. doi: <pub-id pub-id-type="doi">10.1016/j.compag.2020.105220.</pub-id></mixed-citation></ref>
<ref id="ref-48"><label>[48]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. J.</given-names> <surname>Jia</surname></string-name>, <string-name><given-names>P. Y.</given-names> <surname>Jia</surname></string-name>, <string-name><given-names>S. P.</given-names> <surname>Hu</surname></string-name>, and <string-name><given-names>H. B.</given-names> <surname>Liu</surname></string-name></person-group>, &#x201C;<article-title>Automatic detection of tomato diseases and pests based on leaf images</article-title>,&#x201D; in <source>Proc. 2017 Chinese Autom. Congr. (CAC)</source>, <year>2017</year>, pp. <fpage>2510</fpage>&#x2013;<lpage>2537</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CAC.2017.8243388.</pub-id></mixed-citation></ref>
<ref id="ref-49"><label>[49]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Dosovitskiy</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>An image is worth 16 &#x00D7; 16 words: Transformers for image recognition at scale</article-title>,&#x201D; <comment>arXiv:2010.11929</comment>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-50"><label>[50]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>He</surname></string-name>, <string-name><given-names>X.</given-names> <surname>Zhang</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Ren</surname></string-name>, and <string-name><given-names>J.</given-names> <surname>Sun</surname></string-name></person-group>, &#x201C;<article-title>Deep residual learning for image recognition</article-title>,&#x201D; in <source>Proc IEEE Comput. Soc. Conf. Comput. Vis. Pattern Recognit</source>, <year>2016</year>, pp. <fpage>770</fpage>&#x2013;<lpage>778</lpage>. doi: <pub-id pub-id-type="doi">10.1109/CVPR.2016.90.</pub-id></mixed-citation></ref>
<ref id="ref-51"><label>[51]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. L.</given-names> <surname>Gleason</surname></string-name> and <string-name><given-names>B. A.</given-names> <surname>Edmunds</surname></string-name></person-group>, &#x201C;<article-title>Tomato diseases and disorders</article-title>,&#x201D; <publisher-name>Instructional Technology Center, Iowa State University</publisher-name>, pp. <fpage>1</fpage>&#x2013;<lpage>12</lpage>, <year>2006</year>, <comment>Accessed: Mar. 26, 2024</comment>. [Online]. Available: <ext-link ext-link-type="uri" xlink:href="https://ncmg.ucanr.org/files/180088.pdf">https://ncmg.ucanr.org/files/180088.pdf</ext-link>.</mixed-citation></ref>
<ref id="ref-52"><label>[52]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Kaur</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Pandey</surname></string-name>, and <string-name><given-names>S.</given-names> <surname>Goel</surname></string-name></person-group>, &#x201C;<article-title>Plants disease identification and classification through leaf images: A survey</article-title>,&#x201D; <source>Arch. Comput. Methods Eng.</source>, vol. <volume>26</volume>, no. <issue>2</issue>, pp. <fpage>507</fpage>&#x2013;<lpage>530</lpage>, <year>2019</year>. doi: <pub-id pub-id-type="doi">10.1007/s11831-018-9255-6.</pub-id></mixed-citation></ref>
<ref id="ref-53"><label>[53]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Waghmare</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Kokare</surname></string-name>, and <string-name><given-names>Y.</given-names> <surname>Dandawate</surname></string-name></person-group>, &#x201C;<article-title>Detection and classification of diseases of Grape plant using opposite colour Local Binary Pattern feature and machine learning for automated Decision Support System</article-title>,&#x201D; in <source>2016 3rd Int. Conf. Signal Process. Integr. Netw. (SPIN)</source>, <publisher-loc>Noida, India</publisher-loc>, <year>2016</year>, pp. <fpage>513</fpage>&#x2013;<lpage>518</lpage>. doi: <pub-id pub-id-type="doi">10.1109/SPIN.2016.7566749.</pub-id></mixed-citation></ref>
<ref id="ref-54"><label>[54]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Hossain </surname></string-name><etal>et al</etal>.</person-group>, &#x201C;<article-title>Aggregating different scales of attention on feature variants for tomato leaf disease diagnosis from image data: A transformer driven study</article-title>,&#x201D; <source>Sensors</source>, vol. <volume>23</volume>, no. <issue>7</issue>, pp. <fpage>3751</fpage>, <year>Apr. 2023</year>. doi: <pub-id pub-id-type="doi">10.3390/S23073751.</pub-id>; <pub-id pub-id-type="pmid">37050811</pub-id></mixed-citation></ref>
</ref-list>
</back></article>