<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMES</journal-id>
<journal-id journal-id-type="nlm-ta">CMES</journal-id>
<journal-id journal-id-type="publisher-id">CMES</journal-id>
<journal-title-group>
<journal-title>Computer Modeling in Engineering &#x0026; Sciences</journal-title>
</journal-title-group>
<issn pub-type="epub">1526-1506</issn>
<issn pub-type="ppub">1526-1492</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">64395</article-id>
<article-id pub-id-type="doi">10.32604/cmes.2025.064395</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Attention Driven YOLOv5 Network for Enhanced Landslide Detection Using Satellite Imagery of Complex Terrain</article-title>
<alt-title alt-title-type="left-running-head">Attention Driven YOLOv5 Network for Enhanced Landslide Detection Using Satellite Imagery of Complex Terrain</alt-title>
<alt-title alt-title-type="right-running-head">Attention Driven YOLOv5 Network for Enhanced Landslide Detection Using Satellite Imagery of Complex Terrain</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Chandra</surname><given-names>Naveen</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Vaidya</surname><given-names>Himadri</given-names></name><xref ref-type="aff" rid="aff-2">2</xref><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Sawant</surname><given-names>Suraj</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Gite</surname><given-names>Shilpa</given-names></name><xref ref-type="aff" rid="aff-5">5</xref><xref ref-type="aff" rid="aff-6">6</xref></contrib>
<contrib id="author-5" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Pradhan</surname><given-names>Biswajeet</given-names></name><xref ref-type="aff" rid="aff-7">7</xref><email>biswajeet.pradhan@uts.edu.au</email></contrib>
<aff id="aff-1"><label>1</label><institution>Wadia Institute of Himalayan Geology</institution>, <addr-line>Dehradun, 248001</addr-line>, <country>India</country></aff>
<aff id="aff-2"><label>2</label><institution>Graphic Era Hill University</institution>, <addr-line>Dehradun, 248001</addr-line>, <country>India</country></aff>
<aff id="aff-3"><label>3</label><institution>Graphic Era Deemed University</institution>, <addr-line>Dehradun, 248001</addr-line>, <country>India</country></aff>
<aff id="aff-4"><label>4</label><institution>COEP Technological University</institution>, <addr-line>Pune, 411005</addr-line>, <country>India</country></aff>
<aff id="aff-5"><label>5</label><institution>Artificial Intelligence and Machine Learning Department, Symbiosis Institute of Technology, Symbiosis International (Deemed) University</institution>, <addr-line>Pune, 412115</addr-line>, <country>India</country></aff>
<aff id="aff-6"><label>6</label><institution>Symbiosis Centre of Applied AI (SCAAI), Symbiosis International (Deemed) University</institution>, <addr-line>Pune, 412115</addr-line>, <country>India</country></aff>
<aff id="aff-7"><label>7</label><institution>Centre for Advanced Modelling and Geospatial Information Systems (CAMGIS), School of Civil and Environmental Engineering, University of Technology Sydney</institution>, <country>Ultimo, NSW 2007, Australia</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Biswajeet Pradhan. Email: <email>biswajeet.pradhan@uts.edu.au</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>30</day><month>06</month><year>2025</year>
</pub-date>
<volume>143</volume>
<issue>3</issue>
<fpage>3351</fpage>
<lpage>3375</lpage>
<history>
<date date-type="received">
<day>14</day>
<month>2</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>27</day>
<month>5</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMES_64395.pdf"></self-uri>
<abstract>
<p>Landslide hazard detection is a prevalent problem in remote sensing studies, particularly with the technological advancement of computer vision. With the continuous and exceptional growth of the computational environment, the manual and partially automated procedure of landslide detection from remotely sensed images has shifted toward automatic methods with deep learning. Furthermore, attention models, driven by human visual procedures, have become vital in natural hazard-related studies. Hence, this paper proposes an enhanced YOLOv5 (You Only Look Once version 5) network for improved satellite-based landslide detection, embedded with two popular attention modules: CBAM (Convolutional Block Attention Module) and ECA (Efficient Channel Attention). These attention mechanisms are incorporated into the backbone and neck of the YOLOv5 architecture, distinctly, and evaluated across three YOLOv5 variants: nano (n), small (s), and medium (m). The experiments use open-source satellite images from three distinct regions with complex terrain. The standard metrics, including F-score, precision, recall, and mean average precision (mAP), are computed for quantitative assessment. The YOLOv5n &#x002B; CBAM demonstrates the most optimal results with an F-score of 77.2%, confirming its effectiveness. The suggested attention-driven architecture augments detection accuracy, supporting post-landslide event assessment and recovery.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Attention mechanism</kwd>
<kwd>convolutional neural networks</kwd>
<kwd>landslides</kwd>
<kwd>remote sensing images</kwd>
<kwd>YOLOv5</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>University of Technology Sydney</funding-source>
<award-id>EEQ/2022/000812</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Landslides are critical geo-hazards driven by climatic, tectonic, and human-induced activities, causing loss of life, damage to infrastructure and property, and economic disruption [<xref ref-type="bibr" rid="ref-1">1</xref>]. Given current weather patterns, urban expansion, and population growth, more landslides are anticipated in mountainous regions worldwide [<xref ref-type="bibr" rid="ref-2">2</xref>]. Therefore, effective mitigation and associated risk reduction are highly imperative for humanity. Monitoring, prediction/forecasting, and localization/detection are fundamental components in managing landslide risk [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-3">3</xref>,<xref ref-type="bibr" rid="ref-4">4</xref>]. Monitoring involves observing displacement or deformation in an area over an extended period, forecasting refers to predicting the future based on long-term past information, and detection involves extracting information about the occurrence of a landslide [<xref ref-type="bibr" rid="ref-1">1</xref>]. Continuous monitoring of landslides is vital for their prediction, whereas detection leads to the identification of critical parameters, for example, site/location, which are necessary to minimize their cascading effects, aiding first responders during the post-event recovery phase.</p>
<p>The ability to detect landslides quickly after they occur is crucial for initiating immediate response measures. Rapid detection allows emergency responders to deploy quickly rescue teams, medical aid, and supplies to affected areas. It supports the prompt evacuation of people from hazardous zones to prevent further casualties. It aids in performing preliminary assessments of the extent of damage to infrastructure and the environment, which is essential for planning effective recovery operations. Accuracy in landslide detection is vital to ensure that the information used for decision-making is reliable. Accurate detection involves identifying true landslide events by minimizing false positives (incorrectly identified landslides) and false negatives (missed landslides) to provide a clear and precise mapping of affected areas. Accurate detection is crucial for detailed mapping, indicating the precise locations and extent of landslides, which are critical for targeted interventions. Furthermore, timeliness refers to the promptness with which data is gathered, processed, and made available to decision-makers and emergency responders. Timely data is essential as it supports immediate response efforts, reducing the time between the occurrence of a landslide and the initiation of rescue and recovery operations. It also provides real-time situational awareness, enabling responders to understand the current state of affected areas and make informed decisions. In addition, reliability pertains to the data&#x2019;s accuracy and consistency. Decision-makers rely on reliable and dependable data to formulate effective response strategies and allocate resources efficiently. It aids in comprehensive risk assessment, helping to identify high-risk areas and prioritize interventions. Reliable data also supports long-term planning and mitigation efforts, contributing to the development of resilient infrastructure and communities that are better prepared for future landslide events.</p>
<p>Moreover, detecting landslide information is important for post-event actions and inventory preparation. In addition, landslide event detection supports all the later stages of the landslide risk cycle, such as susceptibility assessment, hazard, vulnerability, and risk assessment, which are crucial for sustainable planning. Therefore, fast and accurate detection of landslide events has emerged as a recent trend in the landslide domain. The traditional landslide extraction approach was based on <italic>in-situ</italic> visits [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-6">6</xref>]. These methods were time-consuming, laborious, and less efficient in emergency conditions [<xref ref-type="bibr" rid="ref-6">6</xref>]. With the progress in remote sensing techniques, high-resolution imagery has been widely used in the studies of landslide hazard analysis [<xref ref-type="bibr" rid="ref-1">1</xref>]. Currently, four types of techniques are used for landslide detection employing remote sensing data [<xref ref-type="bibr" rid="ref-5">5</xref>]: visual/manual interpretation and analysis, pixel-based, object-based/slope unit, and artificial intelligence (AI).</p>
<p>Visual interpretation methods are based on the knowledge of experts, nevertheless, they require a lot of time, and struggle to meet the standards necessary for fast recovery actions [<xref ref-type="bibr" rid="ref-6">6</xref>]. Pixel-based methods overcome the limitation of visual methods by using a binary algorithm [<xref ref-type="bibr" rid="ref-5">5</xref>] to recognize the correct class or category (landslide and background) of the pixel in an image. If the objects in an image possess spectral features similar to landslides, it can indeed be challenging to accurately distinguish between them. The object-based methods are based on multi-scale or multi-resolution segmentations that use image primitives/features like shape, spectrum, and texture [<xref ref-type="bibr" rid="ref-7">7</xref>]. These techniques consider threshold criteria and hence require appropriate empirical settings for correct results. In addition, the rapid segmentation of high-resolution satellites across large geographical areas is a tough process [<xref ref-type="bibr" rid="ref-8">8</xref>,<xref ref-type="bibr" rid="ref-9">9</xref>]. A comparative study aiming to explore the strengths and weaknesses of pixel and object-based methods for landslide detection has also been suggested in the past [<xref ref-type="bibr" rid="ref-10">10</xref>].</p>
<p>Furthermore, AI methodologies have undergone significant evolution, substantially impacting various domains, including hazard analysis. The emergence of remote sensing-oriented big data has notably enhanced AI&#x2019;s capabilities [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-12">12</xref>]. Machine learning, a sub-field of AI, has shown considerable progress in landslide prevention and assessment [<xref ref-type="bibr" rid="ref-13">13</xref>]. Notably, algorithms such as support vector machines [<xref ref-type="bibr" rid="ref-13">13</xref>&#x2013;<xref ref-type="bibr" rid="ref-15">15</xref>], logistic regression [<xref ref-type="bibr" rid="ref-16">16</xref>], random forest [<xref ref-type="bibr" rid="ref-17">17</xref>], and decision trees [<xref ref-type="bibr" rid="ref-18">18</xref>] have been applied successfully. While these methods improve efficiency, they often face challenges with accuracy in complex terrains and may require extensive preprocessing. Recent advances in computational technologies have propelled deep learning (a subset of machine learning) into the spotlight, particularly in applications requiring image segmentation [<xref ref-type="bibr" rid="ref-19">19</xref>], object detection [<xref ref-type="bibr" rid="ref-20">20</xref>], and classification [<xref ref-type="bibr" rid="ref-21">21</xref>]. Deep learning has also been increasingly applied in geohazards assessment [<xref ref-type="bibr" rid="ref-4">4</xref>], encompassing earthquakes [<xref ref-type="bibr" rid="ref-22">22</xref>], floods [<xref ref-type="bibr" rid="ref-23">23</xref>], and landslides [<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-25">25</xref>].</p>
<sec id="s1_1">
<label>1.1</label>
<title>Related Work</title>
<p>Diverse architectures of CNNs have been deployed for the detection of landslide-related information [<xref ref-type="bibr" rid="ref-3">3</xref>,<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-24">24</xref>,<xref ref-type="bibr" rid="ref-26">26</xref>&#x2013;<xref ref-type="bibr" rid="ref-28">28</xref>], with notable examples including U-Net [<xref ref-type="bibr" rid="ref-29">29</xref>&#x2013;<xref ref-type="bibr" rid="ref-33">33</xref>], ResNet [<xref ref-type="bibr" rid="ref-34">34</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>], and Mask-RCNN [<xref ref-type="bibr" rid="ref-36">36</xref>,<xref ref-type="bibr" rid="ref-37">37</xref>]. These models use image patches of multiple sizes for binary segmentation, i.e., landslide and background, yet they face notable challenges. The model by Bragagnolo et al. (2021) struggles to accurately detect landslides in complex terrains where visual features resemble surrounding landscapes, making it difficult to distinguish landslides from other elements [<xref ref-type="bibr" rid="ref-29">29</xref>]. Similarly, Ghorbanzadeh et al. (2021) find it challenging to differentiate between landslides and visually similar bare land in optical imagery, suggesting that SAR data could enhance accuracy by providing additional insights [<xref ref-type="bibr" rid="ref-30">30</xref>]. Devara et al. (2024) highlight the dependence on high-quality training data and the need for specific threshold settings, adding complexity to the detection process [<xref ref-type="bibr" rid="ref-33">33</xref>]. Other studies [<xref ref-type="bibr" rid="ref-2">2</xref>,<xref ref-type="bibr" rid="ref-36">36</xref>,<xref ref-type="bibr" rid="ref-37">37</xref>] have evaluated models using UAV-acquired images, Liu et al. (2024) emphasize the importance of empirical settings for better results when dealing with images of varying sizes [<xref ref-type="bibr" rid="ref-37">37</xref>].</p>
<p>YOLO models [<xref ref-type="bibr" rid="ref-38">38</xref>] specifically, YOLOv3 [<xref ref-type="bibr" rid="ref-39">39</xref>,<xref ref-type="bibr" rid="ref-40">40</xref>], YOLOv4 [<xref ref-type="bibr" rid="ref-41">41</xref>], YOLOv5 [<xref ref-type="bibr" rid="ref-42">42</xref>], YOLOv6 [<xref ref-type="bibr" rid="ref-43">43</xref>], YOLOv7 [<xref ref-type="bibr" rid="ref-44">44</xref>], YOLOv8 [<xref ref-type="bibr" rid="ref-45">45</xref>], and YOLOX [<xref ref-type="bibr" rid="ref-46">46</xref>] have been applied by researchers with significant success. To further refine landslide detection accuracy, attention models have been combined with CNNs [<xref ref-type="bibr" rid="ref-5">5</xref>]. These include 3D-SCAM [<xref ref-type="bibr" rid="ref-47">47</xref>], U-Net &#x002B; CBAM [<xref ref-type="bibr" rid="ref-48">48</xref>], LA-YOLO-LLL [<xref ref-type="bibr" rid="ref-49">49</xref>], LD-YOLO: based on ECA [<xref ref-type="bibr" rid="ref-50">50</xref>], YOLO-SA [<xref ref-type="bibr" rid="ref-51">51</xref>], SW-MSA [<xref ref-type="bibr" rid="ref-52">52</xref>], YOLOv7-SE [<xref ref-type="bibr" rid="ref-53">53</xref>], LS-YOLO [<xref ref-type="bibr" rid="ref-54">54</xref>], and LA-YOLO (an improved YOLOv8) [<xref ref-type="bibr" rid="ref-55">55</xref>].</p>
</sec>
<sec id="s1_2">
<label>1.2</label>
<title>Research Gap and Contribution</title>
<p>Although previous studies have laid a significant foundation, several research gaps remain unsolved. Firstly, YOLOv5 has been extensively applied in other object detection tasks, however, there is limited documentation on its application for landslide detection, particularly in complex terrains using remote sensing data. Existing YOLOv5 studies may lack a thorough evaluation of its performance in detecting landslides under challenging conditions such as varying land cover and cloud cover. Secondly, the efficacy of attention mechanisms within single-stage detection algorithms warrants comprehensive evaluation. Previous efforts integrating attention mechanisms with YOLOv5 often involved altering backbone architectures, making it difficult to isolate and evaluate the direct impact of attention mechanisms on performance. This indicates that the true potential of attention models was not fully explored. Thirdly, identifying an appropriate deep-learning model for accurate landslide detection is imperative. Our study addresses these gaps by adding attention mechanisms within the YOLOv5 architecture without altering other components. This allows us to isolate and fully assess how these mechanisms enhance detection accuracy, particularly in challenging landslide detection tasks where distinguishing landslides from complex backgrounds is crucial. By focusing on relevant image features, attention mechanisms improve the model&#x2019;s precision in detecting landslides through remote sensing imagery, and our work demonstrates these advances through comprehensive experimentation.</p>
<p>Landslides often occur suddenly and have severe impacts, necessitating timely and precise information for effective emergency actions. Hence, our work contributes to the ongoing development of algorithms that can support real-time emergency response efforts. While other models have shown potential for landslide detection, our research focuses on improving the precision and efficiency of detecting landslides within image patches, utilizing the YOLOv5 &#x002B; attention model. This specificity in improving detection in challenging scenarios is the unique contribution of our work. By addressing these more detailed aspects of algorithm improvement, our study contributes to the broader goal of rapid and accurate landslide detection, a crucial component of the emergency response framework.</p>
<p>The novel contributions of our work are threefold:
<list list-type="order">
<list-item>
<p>The development and optimization of improved YOLOv5 tailored for automated landslide event detection. We enhance YOLOv5 by refining its architecture to improve feature extraction and classification accuracy in landslide-prone regions.</p></list-item>
<list-item>
<p>The integration of cognitive capabilities through attention mechanisms, specifically CBAM and ECA, into the YOLOv5 architecture to improve hazard scenarios. We embed attention modules within the YOLOv5 backbone to refine spatial and channel-wise feature representation, emphasizing discriminative landslide-related patterns.</p></list-item>
<list-item>
<p>A comprehensive assessment of the model, benchmarked against satellite datasets encompassing diverse geomorphological characteristics. The results validate the effectiveness of our approach, demonstrating superior performance over conventional methods.</p></list-item>
</list></p>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>Study Area and Data Set Description</title>
<p>The research encompasses study sites within Beichuan County in Sichuan Province, Bijie City (Guizhou Province), and Ludian County in Yunnan Province. These areas are recognized for their susceptibility to natural disasters, particularly earthquakes and landslides, which have historically resulted in extensive infrastructural damage [<xref ref-type="bibr" rid="ref-5">5</xref>]. The selected locations are characterized by high mountain ranges, expansive terrain, and dense river networks. The co-seismic landslide events in these regions have inflicted significant damage on essential infrastructure, including tunnels, reservoirs, bridges, roads, and agricultural lands [<xref ref-type="bibr" rid="ref-5">5</xref>]. For this study, a landslide dataset comprising three-channel (RGB) images was acquired from TripleSat (spatial resolution &#x003D; 0.8 m) for Bijie City [<xref ref-type="bibr" rid="ref-47">47</xref>]), covering an area of 26,853 km<sup>2</sup>, captured from May to August 2018 while imagery for Beichuan and Ludian Counties was obtained from the 91 wemap platform [<xref ref-type="bibr" rid="ref-5">5</xref>]. This database comprises a total of 950 images, each with a resolution of 512 &#x00D7; 512 pixels, accompanied by reference data. <xref ref-type="fig" rid="fig-1">Fig. 1</xref> exhibits sample images from the dataset, showcasing the diverse landforms, such as vegetative cover, roadway networks, built-up areas, and aquatic features. The presence of cloud cover within the dataset presents additional complexity, making it particularly challenging for landslide detection algorithms. The dataset can be downloaded from <ext-link ext-link-type="uri" xlink:href="https://github.com/Abbott-max/dataset">https://github.com/Abbott-max/dataset</ext-link> (accessed on 01 January 2025).</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Examples of the dataset used for experiments [<xref ref-type="bibr" rid="ref-5">5</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_64395-fig-1.tif"/>
</fig>
</sec>
<sec id="s3">
<label>3</label>
<title>Methodology</title>
<p>The methodology proposed in this work delineates a systematic procedure for automated landslide detection utilizing satellite imagery. The approach is structured into four primary stages: (i) data annotation and preparation, (ii) model development and architectural design, (iii) training procedures, and (iv) evaluation metrics. The procedural framework of the proposed attention-enhanced YOLOv5 model is encapsulated in Algorithm 1.</p>
<fig id="fig-8">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_64395-fig-8.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Input Data, Annotation, and Formation</title>
<p>The input datasets for our study consist primarily of high-resolution satellite images described in <xref ref-type="sec" rid="s2">Section 2</xref>. The initial annotations of the dataset are represented in the form of polygons (file_type &#x003D; xml). Therefore, we transformed the data into the required format, i.e., bounding boxes, comprising five parameters illustrated in Algorithm 2.</p>
<fig id="fig-9">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_64395-fig-9.tif"/>
</fig>
<p>Additionally, a YAML file (a standard YOLO format), containing necessary information such as the file paths for training, validation, and testing datasets, the number of classes, and the class names (landslide and background), has been created. To facilitate experimentation, the annotated data has been divided into three subsets: 70% for training, 20% for validation, and 10% for testing. In our study, we used the Roboflow-guided approach for this data distribution. Roboflow helps manage and prepare datasets for machine learning by automatically partitioning them into balanced training, validation, and testing subsets. This ensures diverse and representative data distribution, reducing bias and improving the model&#x2019;s accuracy and generalization for landslide detection.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Model Development and Architecture</title>
<p>This section explains the baseline model (YOLOv5), the implemented attention modules (CBAM and ECA), and the proposed network in detail.</p>
<sec id="s3_2_1">
<label>3.2.1</label>
<title>Description of the Original YOLOv5 Network</title>
<p>YOLO models [<xref ref-type="bibr" rid="ref-56">56</xref>,<xref ref-type="bibr" rid="ref-57">57</xref>] are popular one-staged object detection algorithms comprising three essential units. First, the Backbone, a CNN, constructs image features. Second, the Neck, a series of layers, combines different image features for subsequent predictions. Third, the <bold>Head</bold> accumulates image features from the preceding module (neck), facilitating predictions. YOLOv5, presented by Ultralytics (<ext-link ext-link-type="uri" xlink:href="https://github.com/ultralytics/yolov5">https://github.com/ultralytics/yolov5</ext-link> (accessed on 01 January 2025)), is a prominent addition to the YOLO family.</p>
<p>We select YOLOv5 as our primary model for three main reasons. First, YOLOv5 integrates Cross-Stage Partial Network (CSPNet) within the Darknet, forming CSPDarknet, the backbone [<xref ref-type="bibr" rid="ref-58">58</xref>]. CSPNet addresses issues related to redundant gradient information by incorporating gradient changes into the feature map [<xref ref-type="bibr" rid="ref-58">58</xref>]. This approach reduces the model&#x2019;s parameters and Floating-Point Operations Per Second, ensuring both inference speed and accuracy while minimizing model size as well as the significance compact model size for efficient inference on restricted computational platforms. YOLOv5 with CSPDarknet is a suitable choice.</p>
<p>The main architecture consists of stacking multiple CBS (convolution &#x002B; batch normalization &#x002B; sigmoid linear unit) nodes and concentrated-comprehensive-convolution (C3), modules, with one spatial-pyramid-pooling-fast (SPPF) module connected at the end [<xref ref-type="bibr" rid="ref-59">59</xref>]. Second, YOLOv5 uses the Path Aggregation Network (PANet) [<xref ref-type="bibr" rid="ref-60">60</xref>] to enhance information flow. PANet employs a Feature Pyramid Network (FPN) [<xref ref-type="bibr" rid="ref-61">61</xref>] with an enhanced bottom-up path, facilitating the propagation of low-level features and aiding the model in generalizing effectively to objects of various sizes and scales. These significant characteristics of the neck module of YOLOv5 make it suitable for complex scenarios like landslide detection. Additionally, adaptive feature pooling is utilized [<xref ref-type="bibr" rid="ref-60">60</xref>], connecting the feature grid with all feature levels to enable useful information in each level to propagate directly to subsequent sub-networks. PANet enhances the utilization of precise localization signals in lower layers, thereby significantly refining object location accuracy [<xref ref-type="bibr" rid="ref-60">60</xref>].</p>
<p>Thirdly, the YOLOv5 head generates feature maps of three different sizes to facilitate multi-scale prediction [<xref ref-type="bibr" rid="ref-56">56</xref>]. This feature enables the model to detect objects of varying sizes. For example, in landslide detection tasks, where satellite images may contain landslide events of different sizes (including small, medium, and large), multi-scale detection ensures that the model can accurately identify and extract these events regardless of their scale. Further, data augmentation is a key training process in the YOLOv5 model that introduces transformations to the initial training set, expanding the model&#x2019;s exposure to a broader spectrum of semantic variations beyond the isolated training dataset. In particular, mosaic, translation, scaling, flipping, and HSV augmentation techniques are applied in this study to diversify the training data, ensuring the model&#x2019;s capacity to generalize across a range of inputs [<xref ref-type="bibr" rid="ref-62">62</xref>].</p>
<p>YOLOv5 computes CIoU (complete-intersection-over-union) bounding box regression loss function to optimize the accuracy [<xref ref-type="bibr" rid="ref-63">63</xref>]. CIoU loss incorporates the geometric relationships between predicted and ground truth bounding boxes, leading to better localization accuracy compared to traditional Intersection over Union (IoU). Moreover, CIoU loss considers the aspect ratio and size differences between predicted and ground truth boxes, making it more robust to variations in object size and shape. In addition, CIoU loss effectively accounts for overlapping regions, enabling the model to more accurately distinguish between closely located objects. Compared to the other loss functions, CIoU is less affected by vanishing gradients and saturation, leading to more stable training dynamics and faster convergence [<xref ref-type="bibr" rid="ref-59">59</xref>]. Besides, CIoU loss correlates well with evaluation indicators, ensuring that the loss function aligns with the desired model performance metrics. Furthermore, five versions of YOLOv5 (YOLOv5n, YOLOv5s, YOLOv5m, YOLOv5l, YOLOv5x) are available [<xref ref-type="bibr" rid="ref-57">57</xref>]. These versions are initially trained on the large-scale MS-COCO database, enabling them to acquire rich and generalizable features from a diverse array of images. Therefore, when these pre-trained models are used for applications like landslide detection, where collecting large amounts of labeled data may be challenging, their generalization ability allows them to perform well even when trained on relatively small or limited datasets. Considering the size of the dataset and the existing computing facilities, our study focuses on utilizing YOLOv5n, YOLOv5s, and YOLOv5m models.</p>
</sec>
<sec id="s3_2_2">
<label>3.2.2</label>
<title>CBAM</title>
<p>According to Woo et al. (2018), a CBAM is an effective attention framework that can be fused into deep learning models with remarkably fewer parameters [<xref ref-type="bibr" rid="ref-64">64</xref>]. As seen in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>, the CBAM includes two blocks: (i) a channel attention module (CAM) that undertakes two essential tasks: global average pooling (GAP) followed by maximum global pooling. (ii) A spatial attention module (SAM) that performs maximum as well as average pooling. This architecture enables the computation of attention-based weights for accurate feature map refinement. GAP computes the average value of individual feature maps through all spatial locations, capturing the overall distribution of information, while maximum pooling identifies the most significant features within each feature map by selecting the maximum value. Therefore, the network can capture both the general distribution of features and the most salient features in each map. Further, GAP provides a robust representation of the entire feature map, ensuring that important information is not lost during pooling. Meanwhile, maximum pooling focuses on extracting the most relevant features, enhancing the discriminative power of the pooled representation. Combining these pooling methods creates a comprehensive representation that incorporates global context and local features. The integration of GAP and maximum pooling also allows the network to generalize well to different input variations. GAP provides a stable representation that is less sensitive to small changes in the input, however, maximum pooling focuses on extracting robust features that are invariant to translation or rotation. This combination enriches the network&#x2019;s capability to learn discriminative features across diverse input samples. Hence, combining GAP and maximum pooling enables the network to effectively capture both global context and local characteristics, leading to improved performance in critical tasks like landslide information extraction.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Structural framework of CBAM [<xref ref-type="bibr" rid="ref-60">60</xref>]</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_64395-fig-2.tif"/>
</fig>
<p>Furthermore, the mathematical description of CBAM is as follows: via convolution and pooling procedures, the CBAM calculates the 1-D channel attention (<inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>C</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>) and 2-D spatial attention maps (<inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>) using an input feature map (<inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mi>F</mml:mi><mml:mi>M</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>). The total computed attention is estimated by <xref ref-type="disp-formula" rid="eqn-1">Eqs. (1)</xref> and <xref ref-type="disp-formula" rid="eqn-2">(2)</xref>.
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mi>F</mml:mi><mml:msup><mml:mi>M</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>C</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mi>M</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2297;</mml:mo><mml:mi>F</mml:mi><mml:mi>M</mml:mi></mml:math></disp-formula>
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mi>F</mml:mi><mml:msup><mml:mi>M</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>S</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:msup><mml:mi>M</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2297;</mml:mo><mml:mi>F</mml:mi><mml:msup><mml:mi>M</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:math></disp-formula>where &#x2018;<inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mo>&#x2297;</mml:mo></mml:math></inline-formula>&#x2019; denotes element-wise multiplication and <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mi>F</mml:mi><mml:msup><mml:mi>m</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> signifies the enhanced outcome.</p>
<p>In addition, the weights (CAM and SAM) (Woo et al., 2018; Yang et al., 2022) are computed by <xref ref-type="disp-formula" rid="eqn-3">Eqs. (3)</xref> and <xref ref-type="disp-formula" rid="eqn-4">(4)</xref>.
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>W</mml:mi><mml:mi>e</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>h</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:mi>S</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>M</mml:mi><mml:mi>u</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>y</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>M</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>F</mml:mi><mml:mi>M</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd /><mml:mtd><mml:mi></mml:mi><mml:mo>+</mml:mo><mml:mi>M</mml:mi><mml:mi>u</mml:mi><mml:mi>l</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>y</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>A</mml:mi><mml:mi>v</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>g</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>F</mml:mi><mml:mi>M</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mi>S</mml:mi><mml:mi>A</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>W</mml:mi><mml:mi>e</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>h</mml:mi><mml:mi>t</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>S</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi></mml:math></disp-formula></p>
</sec>
<sec id="s3_2_3">
<label>3.2.3</label>
<title>ECA-Net</title>
<p>ECA-Net [<xref ref-type="bibr" rid="ref-65">65</xref>], a lightweight model comparable to SENet [<xref ref-type="bibr" rid="ref-66">66</xref>]. ECA-Net is designed to be more computationally efficient than SENet. Due to its squeeze-and-excitation operations, SENet introduces additional parameters and computational overhead, which involve learning parameters for each channel and performing GAP operations. The ECA module introduces minimal computational overhead by performing 1D convolution along the channel dimension, resulting in a more efficient architecture. ECA-Net follows sparsity in the connectivity pattern of the attention mechanism to reduce computational complexity further. This sparsity helps limit the number of connections and operations required for feature recalibration, leading to faster inference and reduced memory consumption compared to SENet. The ECA module effectively captures channel-wise dependencies and adaptively recalibrates features, leading to improved feature representations and enhanced model performance. Thus, ECA-Net is well-suited for environments with limited resources and real-time applications, for example, landslide detection, where efficiency is crucial. ECA-Net offers greater flexibility in terms of model architecture and scalability. The output of the convolutional layers is a 4-D tensor, serving as input to ECA-Net. The architecture of ECA-Net comprises three modules [<xref ref-type="bibr" rid="ref-61">61</xref>]: (i) global feature descriptor, (ii) adaptive neighborhood interaction, and (iii) broadcasted scaling. The GAP operation implicates processing the served input tensor by averaging all pixels in a specific feature map, resulting in a single pixel. Subsequently, the tensor <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mrow><mml:mo>(</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, undergoes a 1-D striding convolution with a kernel size denoted as &#x2018;<inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:msub><mml:mi>K</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>&#x2019;. Additionally, the adaptive estimation of &#x2018;<inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:msub><mml:mi>K</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>&#x2019; is determined based on the channel space <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:mrow><mml:mo>(</mml:mo><mml:mi>C</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, utilizing <xref ref-type="disp-formula" rid="eqn-5">Eqs. (5)</xref> and <xref ref-type="disp-formula" rid="eqn-6">(6)</xref>.
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:msub><mml:mi>K</mml:mi><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x03C8;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>C</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo>|</mml:mo><mml:mfrac><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mi>C</mml:mi></mml:mrow><mml:mi>&#x03B3;</mml:mi></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mi>z</mml:mi><mml:mi>&#x03B3;</mml:mi></mml:mfrac><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>d</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula>
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x2205;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>k</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mn>2</mml:mn><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo>&#x2217;</mml:mo><mml:mi>k</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>z</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:math></disp-formula>where <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mi>&#x03B3;</mml:mi></mml:math></inline-formula>, &#x2018;<italic>z</italic>&#x2019; signify the predefined hyper-parameters.</p>
</sec>
<sec id="s3_2_4">
<label>3.2.4</label>
<title>Proposed Attention Embedded YOLOv5 Network</title>
<p>Attention, a primary cognitive capability of humanoids, allows them to process the most pertinent information, eliminating the insignificant information. In this regard, attention modules, inspired by the human vision system, show substantial potential by their integration within a CNN to augment its performance. Hence, we amalgamated two attention mechanisms particularly CBAM and ECA, separately in the backbone and the neck of the YOLOv5 network.</p>
<p>The original YOLOv5 backbone encompasses a sequence of convolutional layers, bottleneck blocks (C3, denoting concentrated-comprehensive-convolution), and SPPF (spatial-pyramid-pooling-fast), systematically downsampling the input image. The backbone is optimized by adding an attention module (especially CBAM and ECA discretely) between the C3 layer and SPPF. This augments the network&#x2019;s ability to capture multi-scale spatial and contextual features before they are fed into the neck for further processing. Subsequently, within the neck, the attention module is integrated after the last three C3 layers for enhancement. This ensures feature refinement before multi-scale feature fusion, improving the ability of the PANet structure to focus on important landslide features.</p>
<p>The architecture of the proposed network is shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>. Embedding attention in YOLOv5n, YOLOv5s, and YOLOv5m networks aids several parameters. First, model depth_multiple, layer channel_multiple, and the total number of classes (01). The model&#x2019;s depth is regulated by the depth_multiple parameter, which scales the number of layers within each module (C3 or Conv). Acting as a multiplier, it plays a key role in determining the overall depth of the network. However, the channel_multiple parameter serves to scale the number of channels in each layer, thereby influencing the model&#x2019;s width. This critical parameter directly impacts the computational settings and the memory requirements, shaping the trade-off between model complexity and resource efficacy. Second, it creates anchor boxes based on various detection scales within the convolutional layer of its backbone. These anchor boxes are vital for ascertaining the dimensions and positions/locations of the detected objects. The addition of an attention module in the final layer of the backbone enhances the YOLOv5 discriminative capabilities. Moreover, the neck defines the construction of the detection head in YOLOv5, comprising convolutional layers, upsampling layers, and concatenation tasks. These elements assist in the integration of feature maps derived from different scales of convolutional layers within the backbone of the proposed network. Further, an attention module is appended at the end of each layer to refine the extracted features. Based on the generated feature maps and anchor boxes, the head (detection module) performs the task of extracting landslide events. The proposed architecture is aimed at enhancing accuracy, enabling real-time extraction of landslide events from remote sensing data.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Proposed YOLOv5 &#x002B; attention-based network for landslide detection</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_64395-fig-3.tif"/>
</fig>
</sec>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Training Setup</title>
<p>During training, an image size of 640 &#x00D7; 640 pixels is served as an input to the proposed network. The choice of a 640 &#x00D7; 640 pixel image size for our landslide detection model is strategically aligned with the spatial resolution of the imagery and the scale of landslide features. This image size is generally sufficient to capture a variety of landslide sizes, ensuring that the model can detect features effectively while maintaining a balance between detail and computational efficiency. Additionally, the YOLOv5 architecture is designed to efficiently process input images of size 640 &#x00D7; 640 pixels. Moreover, the batch size is 8, epochs 500. We used a stochastic gradient descent optimizer with an initial learning rate of 0.01, an initial momentum factor of 0.937, and an initial weight decay of 0.0005. Besides, the considered hyperparameters are based on previous studies [<xref ref-type="bibr" rid="ref-44">44</xref>,<xref ref-type="bibr" rid="ref-46">46</xref>] and directly adopted for the current experiment. Furthermore, the machine configuration includes CPU: Intel&#x00AE; Xeon-CPU E3-1231 (v3@3.40 GHz), GPU (graphics processing unit): NVIDIA-GeForce-RTX3080Ti, deep learning framework: PyTorch 1.7, operating system: Ubuntu 18.04, and CUDA11.4.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Evaluation Criteria</title>
<p>The performance of the YOLOv5 &#x002B; attention model is evaluated using both quantitative and qualitative criteria. The four primary metrics used for quantitative assessment are precision, recall, F-score, and mAP (<xref ref-type="disp-formula" rid="eqn-7">Eqs (7)</xref>&#x2013;<xref ref-type="disp-formula" rid="eqn-10">(10)</xref>). The use of these is well-established and has been extensively used in previous studies [<xref ref-type="bibr" rid="ref-5">5</xref>,<xref ref-type="bibr" rid="ref-48">48</xref>,<xref ref-type="bibr" rid="ref-49">49</xref>,<xref ref-type="bibr" rid="ref-67">67</xref>,<xref ref-type="bibr" rid="ref-68">68</xref>]. Below is a detailed explanation of each metric:
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>(</mml:mo><mml:mi>T</mml:mi><mml:mi>r</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>T</mml:mi><mml:mi>r</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>(</mml:mo><mml:mi>T</mml:mi><mml:mi>r</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mi>T</mml:mi><mml:mi>r</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>e</mml:mi><mml:mi>g</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mi>F</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>S</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mi>m</mml:mi><mml:mi>A</mml:mi><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mfrac><mml:mn>1</mml:mn><mml:mi>n</mml:mi></mml:mfrac><mml:msubsup><mml:mo>&#x222B;</mml:mo><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>On the other hand, qualitative evaluation is conducted through visual analysis of the computed results.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experimental Results</title>
<p>Here, the experimental findings of the proposed network are given. First, the quantitative results derived from the metrics explained in <xref ref-type="sec" rid="s3_4">Section 3.4</xref> are presented. Second, the qualitative outputs of the extracted landslides are described. Third, the comparative analysis of the obtained results is described. Fourth, the overall discussion is mentioned.</p>
<sec id="s4_1">
<label>4.1</label>
<title>Quantitative Evaluation</title>
<p>The results of quantitative indicators, i.e., mAP, precision, recall, and f-score of the attention-embedded YOLOv5 are shown in <xref ref-type="table" rid="table-1">Table 1</xref>. The mAP@0.5 of the YOLOv5 variants (baseline) particularly, YOLOv5n, YOLOv5s, and YOLOv5m is 73.1%, 75.1%, and 72.5%, respectively.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Experimental results of YOLOv5 &#x002B; attention-based models</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="center">Model</th>
<th align="center">GFLOPs</th>
<th align="center">Parameters (m)</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F-Score</th>
<th align="center">mAP@0.5</th>
<th align="center">Time (h)</th>
<th align="center">GPU (GBs)</th>
</tr>
</thead>
<tbody>
<tr>
<td>YOLOv5n</td>
<td>4.1</td>
<td>1.76</td>
<td>71.3</td>
<td>74.5</td>
<td>72.9</td>
<td>73.1</td>
<td>0.54</td>
<td>1.10</td>
</tr>
<tr>
<td>YOLOv5n &#x002B; CBAM</td>
<td>4.3</td>
<td>1.91</td>
<td>84.4</td>
<td>71.1</td>
<td>77.2</td>
<td><bold><italic>77.8</italic></bold></td>
<td>0.88</td>
<td>1.10</td>
</tr>
<tr>
<td>YOLOv5n &#x002B; ECA</td>
<td>8.1</td>
<td>3.91</td>
<td>76.0</td>
<td>75.2</td>
<td>75.6</td>
<td><italic>77.0</italic></td>
<td>1.19</td>
<td>1.93</td>
</tr>
<tr>
<td>YOLOv5s</td>
<td>15.8</td>
<td>7.01</td>
<td>73.3</td>
<td>67.4</td>
<td>70.2</td>
<td>75.1</td>
<td>0.53</td>
<td>1.88</td>
</tr>
<tr>
<td>YOLOv5s &#x002B; CBAM</td>
<td>16.2</td>
<td>7.62</td>
<td>70.4</td>
<td>73.9</td>
<td>72.1</td>
<td>76.1</td>
<td>0.68</td>
<td>2.17</td>
</tr>
<tr>
<td>YOLOv5s &#x002B; ECA</td>
<td>31.5</td>
<td>15.61</td>
<td>80.7</td>
<td>70.5</td>
<td>75.3</td>
<td>76.8</td>
<td>1.53</td>
<td>3.88</td>
</tr>
<tr>
<td>YOLOv5m</td>
<td>47.9</td>
<td>20.85</td>
<td>75.2</td>
<td>64.7</td>
<td>69.6</td>
<td>72.5</td>
<td>0.91</td>
<td>3.73</td>
</tr>
<tr>
<td>YOLOv5m &#x002B; CBAM</td>
<td>49.0</td>
<td>22.21</td>
<td>76.4</td>
<td>75.3</td>
<td>75.8</td>
<td>76.8</td>
<td>1.33</td>
<td>3.85</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The mAP is a comprehensive metric that summarizes the precision-recall trade-off across different confidence thresholds. An increase in mAP indicates an overall improvement in the model&#x2019;s ability, providing a more accurate and reliable detection performance across various conditions. A higher mAP reflects that the model is consistently detecting landslides with both high precision and high recall, making it effective in diverse and challenging environments. Considering the YOLOv5n &#x002B; CBAM model, the mAP@0.5 computed is 77.8%, which showed an improvement of &#x002B;4.7%. Further, an enhancement of &#x002B;1.0% is noticed in mAP@0.5 of YOLOv5s &#x002B; CBAM (76.1%). Moreover, in the case of YOLOv5m &#x002B; CBAM, the estimated progress is &#x002B;4.3% (mAP@0.5). Besides, ECA assimilated YOLOv5 also demonstrated the increased accuracy, mainly YOLOv5n &#x002B; ECA (mAP@0.5 &#x003D; 77.0%) and YOLOv5s &#x002B; ECA (mAP@0.5 &#x003D; 76.8%) unveil the progress of &#x002B;3.9%, and &#x002B;1.7%, respectively. Here, mAP evaluates how well the YOLOv5 model, enhanced with attention mechanisms (CBAM and ECA), detects landslides across a range of confidence thresholds. Unlike precision and recall, which are evaluated at a single threshold, mAP considers the performance at multiple thresholds, providing a holistic view of the model&#x2019;s accuracy. In landslide detection, variations in confidence levels can affect the reliability of detections. mAP provides a comprehensive summary of how well the model performs across these different levels, ensuring that the evaluation is not biased by a single confidence threshold.</p>
<p>In addition, a difference of 1.60% is observed in the f-score of YOLOv5n &#x002B; CBAM and YOLOv5n &#x002B; ECA; however, the improvement reached &#x002B;4.3% and &#x002B;2.7%, respectively. Significant progress of &#x002B;1.9% and &#x002B;5.1% is seen in the f-score of YOLOv5s &#x002B; CBAM and YOLOv5s &#x002B; ECA, respectively. The highest improvement of &#x002B;6.2% in the F-score is obtained in the case of YOLOv5s &#x002B; CBAM. The F-score is a critical metric for evaluating landslide detection models as it balances precision and recall, handles imbalanced datasets, provides a single performance metric for easy comparison, helps interpret model trade-offs, and reflects real-world applicability. For landslide detection using models like YOLOv5 enhanced with attention mechanisms, the F-score ensures that both false positives and false negatives are minimized, leading to more accurate and reliable detection and mapping of landslides. The precision and recall ranged from 70.4% to 84.4% and 64.4% to 75.3%. Precision measures the proportion of correctly identified landslides out of all detected landslides. However, recall estimates the proportion of actual landslides that the model successfully detects. Considering the evaluation metrics, the YOLOv5n &#x002B; CBAM exhibited the best performance.</p>
<p>In landslide detection tasks, calculating GLOPs (Giga Floating Point Operations), the number of parameters, the GPU used, and computational time are essential for estimating the model&#x2019;s performance. GLOPs and parameters provide insights into the model&#x2019;s complexity and computational load, which directly affect its efficiency and feasibility for real-time applications. The computed GLOPs (highest &#x003D; 49.0 (YOLOv5m &#x002B; CBAM), and lowest &#x003D; 4.1 (YOLOv5n)) and the number of parameters (maximum &#x003D; 22.21 (YOLOv5m &#x002B; CBAM), and minimum &#x003D; 1.76 (YOLOv5n)) in millions (m) by the baseline and attention integrated models are also shown in <xref ref-type="table" rid="table-1">Table 1</xref>.</p>

<p>The computational time is an important parameter for practical deployment, especially in disaster response scenarios where timely detection and mapping of landslides are imperative. Together, these metrics help in evaluating the model&#x2019;s performance, optimizing resource allocation, and ensuring that the system can deliver rapid and accurate results, thereby enhancing its applicability and reliability in real-world landslide detection tasks. The execution time of the YOLOv5s &#x002B; ECA network is maximum (1.53 h) while the YOLOv5n &#x002B; CBAM takes the minimum time (0.88 h). The execution time of two baseline models (YOLOv5n &#x003D; 0.54 h and YOLOv5s &#x003D; 0.53 h) is almost similar (difference &#x003D; 0.01 h), whereas YOLOv5m is executed in 0.91 h.</p>
<p>Our study is limited to analyzing landslides within fixed-size image patches, and as such, the total coverage area is not explicitly defined in our current analysis. While this patch-based approach simplifies detailed analysis, it doesn&#x2019;t assess performance over larger, contiguous regions, which would require processing multiple patches for broader applications.</p>
<p>Furthermore, the GPU used is crucial as it determines the model&#x2019;s processing capability and speed; different GPUs offer varying levels of performance that can significantly impact the model&#x2019;s ability to handle large datasets quickly. The GPU or memory utilization (GBs) by each model is also given in <xref ref-type="table" rid="table-1">Table 1</xref>. It is observed that GPU consumption is directly related to the size of the model (YOLOv5n &#x003D; 1.10 GBs, YOLOv5s &#x003D; 1.88 GBs, and YOLOv5m &#x003D; 3.73 GBs). Moreover, estimating the loss functions is crucial for the proposed landslide detection. The bounding box regression loss computes how accurately the predicted bounding boxes align with the actual landslide locations, ensuring that the model precisely identifies the spatial extent of landslides. Confidence loss, on the other hand, evaluates the model&#x2019;s certainty in its predictions, distinguishing between true landslide detections and false positives.</p>

<p>The assessed loss functions (training and validation) of the five proposed networks (YOLOv5n &#x002B; CBAM, YOLOv5s &#x002B; CBAM, YOLOv5m &#x002B; CBAM, YOLOv5n &#x002B; ECA, and YOLOv5s &#x002B; ECA) are presented in <xref ref-type="fig" rid="fig-4">Fig. 4a</xref>,<xref ref-type="fig" rid="fig-4">b</xref>, respectively. When the bounding box regression loss is low, it indicates that the model is accurately predicting the locations of the landslides with minimal error, i.e., the predicted boxes are close to the actual ground truth boxes. However, a high bounding box regression loss suggests that the predicted bounding boxes are not aligned well with the true locations of the landslides. This could indicate that the model struggles to localize the landslide accurately, resulting in poor detection performance. Further, confidence loss estimates how assured the model is in its prediction that a detected object is indeed a landslide. A low confidence loss means the model is effectively distinguishing between landslides and non-landslide regions. The model is confident when it makes correct detections and assigns high confidence scores to these predictions. Higher confidence loss shows that the model is either missing some landslides (false negatives) or incorrectly identifying non-landslide regions as landslides (false positives). This reflects a lack of confidence in making accurate predictions. Hence, the lower function represents that the model is performing well (YOLOv5n &#x002B; CBAM). It can accurately locate and identify landslides with high confidence, meaning that both the bounding box and object classification tasks are being handled correctly. On the other hand, a higher loss function signifies that the model needs further training or optimization. Either it struggles with localizing landslides, identifying them, or both, leading to errors.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title><bold>(a)</bold>. Training loss function estimation. <bold>(b)</bold>. Validation loss function estimation</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_64395-fig-4.tif"/>
</fig>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Qualitative Evaluation</title>
<p>The visual outcomes of the extracted landslides by YOLO plus attention models are represented by bounding boxes shown in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>. The bounding box is a standard in object detection frameworks, especially those based on the YOLO architecture. Bounding boxes provide a simplified and efficient means of detecting and localizing objects within an image. The primary goal of the YOLOv5 &#x002B; attention model is to quickly identify potential landslide areas from satellite imagery. Bounding boxes allow for rapid detection, which is crucial for time-sensitive applications such as post-disaster response. Moreover, training a model to detect and delineate exact polygonal outlines is significantly more complex and computationally intensive compared to detecting bounding boxes. Bounding boxes serve as a practical compromise that enables efficient training and inference while still providing valuable spatial information about the landslides. The use of bounding boxes allows the model to scale more effectively when processing large geographical areas. Given the extensive size of satellite images and the potential number of landslide events to be detected, bounding boxes enable the model to process images more quickly and with fewer computational resources. While bounding boxes provide a useful starting point for landslide detection, we acknowledge that detailed polygonal outlines are more desirable for many applications, such as accurate mapping and geospatial analysis. Irrespective of the spatial location, the landslides of small (A1), as well as moderate (B1) sizes in an image, are successfully extracted by the YOLOv5n &#x002B; CBAM model. The model can differentiate the landslides of different extents. The dataset comprises images with complex backgrounds, which makes it challenging for landslide detection. For example, the presence of clouds, and landscapes with homogenous characteristics (landslides, and bare sand). In <xref ref-type="fig" rid="fig-5">Fig. 5</xref> C1, some of the landslides are difficult to interpret manually, however model predicts them effectively. The images in column 4 (D1) indicate the examples where the models struggle to detect landslide events. Nevertheless, YOLOv5n &#x002B; CBAM showcased high detection precision and robustness. Hence, automated landslide recognition models are anticipated for better prediction results.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Qualitative results of the extracted landslides by YOLOv5n &#x002B; CBAM (Labels: A1, B1, C1, and D1, detected landslides: A2, B2, C2, and D2)</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_64395-fig-5.tif"/>
</fig>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Comparison with the Advanced YOLO Models and Previous Work</title>
<p>For the considered dataset, the competency of the advanced YOLO models specifically, YOLOv6 (n, s, and m), YOLOv7 tiny, and YOLOv8 (n, and s) is evaluated. Moreover, YOLO-NAS (s, and m), a recent object detection model, is also executed to examine its capability in the studies of landslide hazards. In total, we compared the obtained results with eight state-of-the-art YOLO models. The obtained mAP@0.5 of all the models (<xref ref-type="fig" rid="fig-6">Fig. 6</xref>). The results indicate that the maximum accuracy is achieved by YOLOv8s (74.8%) followed by YOLOv8n (73.4%). The performance of YOLO-NASm (60.4%) is the least; however, a notable improvement is seen in YOLO-NASs (65.4%). A difference of 1.2% is observed in the case of YOLOv6s (67.3%), and YOLOv6m (68.5%), whereas almost similar accuracy is estimated for YOLOv6n (69.7%), and YOLOv7 tiny (69.0%) models. The significant improvement of &#x002B;3.0% to &#x002B;17.4%, and &#x002B;2.2% to &#x002B;16.6% in the mAP@0.50 of CBAM, and ECA-integrated YOLOv5 networks, respectively demonstrates the superior performance of our approach compared to other state-of-the-art YOLO models. Our model not only outperforms earlier YOLO versions but even more advanced models such as YOLOv8 and YOLO-NAS. This comparison and wide range of improvement prove the robustness of our attention-enhanced architecture across challenging environments.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Performance of the advanced YOLO models using satellite images</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_64395-fig-6.tif"/>
</fig>
<p>Further, the lack of direct comparison with previous studies in our work is primarily due to the unique nature of our dataset and the novel combination of models we employed. The dataset used in our study has not been extensively utilized in previous studies. This uniqueness makes direct comparison challenging. Earlier studies on landslide detection have used different datasets, making direct performance comparison unfeasible. A justified comparison can only be made with studies using the same data and models. Our study aims to establish a baseline for future research using this specific dataset and methodology. However, we compared our results with the existing YOLOv5 &#x002B; attention models applied to other datasets with dissimilar characteristics. The comparison between the suggested YOLOv5n &#x002B; CBAM, YOLOv5n &#x002B; ECA models and earlier studies, particularly LP-YOLO [<xref ref-type="bibr" rid="ref-44">44</xref>] and YOLOv5 &#x002B; ASSF &#x002B; CBAM [<xref ref-type="bibr" rid="ref-69">69</xref>], highlights the important advancements made in terms of performance metrics, specifically mAP@0.5, precision, and recall. Compared to LP-YOLO, our YOLOv5n &#x002B; CBAM model shows a significant improvement in mAP@0.5 (&#x002B;28.8%), precision (&#x002B;30.7%), and recall (&#x002B;21.3%). This shows that our model is enhanced at correctly identifying landslides, minimizing both false positives and false negatives, which is critical in applications requiring high accuracy. Moreover, YOLOv5n &#x002B; ECA also shows a noteworthy improvement over LP-YOLO, with increases in mAP@0.5 (&#x002B;28.0%), precision (&#x002B;22.3%), and recall (&#x002B;25.4%). These metrics demonstrate that the ECA-based model also yields challenging results.</p>
<p>In comparison with the YOLOv5 &#x002B; ASSF &#x002B; CBAM model [<xref ref-type="bibr" rid="ref-64">64</xref>], our YOLOv5n &#x002B; CBAM shows a &#x002B;3.8% increase in mAP@0.5 and an improvement of &#x002B;6.0% in precision, although no progress is observed in recall. This illustrates that our model not only matches existing attention-based approaches but even exceeds them in terms of accuracy and precision, making it a stronger alternative. In addition, YOLOv5n &#x002B; ECA also demonstrates an improvement of &#x002B;3.0% in mAP@0.5, further showcasing how incorporating different attention mechanisms can result in improvements over previously existing models.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Discussion</title>
<p>The proposed model is a trained YOLOv5 model enhanced with attention mechanisms such as CBAM and ECA. This advanced model is specifically designed for the detection and mapping of landslides using remote sensing imagery. The model&#x2019;s training process involves several critical phases, which include the acquisition and annotation of training data, the integration of attention mechanisms during model training, and the fine-tuning and validation of the model to ensure optimal performance. These steps are essential for developing a model that can accurately identify landslide features in input images and generate detailed, reliable outputs. The timescales for data generation are largely dependent on the size and complexity of the dataset; however, once trained, the model is capable of processing and generating maps for small to medium-sized areas efficiently. The minimum mappable feature size is a critical factor in the effectiveness of our YOLOv5 &#x002B; attention model for landslide detection. The model&#x2019;s ability to identify and map small landslide features is constrained by the resolution of the input images and the model&#x2019;s detection capabilities. Regarding input requirements, the model needs high-resolution satellite images to effectively detect landslides, with multi-spectral images being preferred due to their ability to capture detailed features. Additionally, high-resolution UAV images can be utilized for detailed analyses of smaller areas. The output from the model is presented as bounding boxes that accurately indicate the presence of landslides, aiming for high spatial accuracy with minimal misalignment. This ensures that the detected landslides closely match their actual locations.</p>
<p>Moreover, a relatively lower recall of the YOLOv5n &#x002B; CBAM model is noted, despite its high precision and F-score. This indicates that while the model effectively minimizes false positives, it may overlook some actual landslide occurrences. To address this, we analyzed potential reasons for the lower recall. Filtering of subtle features: The CBAM module enhances feature selection but might suppress small-scale landslides. Detection threshold: A high confidence threshold may lead to fewer false positives but could also exclude some true positives. Dataset characteristics: The presence of complex terrain and vegetation may introduce challenges in detecting all landslides. Spatial resolution constraints: Smaller landslides may not be well represented due to the image resolution limitations. To improve recall while maintaining high precision, we plan to explore several approaches. Fine-tuning the detection confidence threshold could help retain more potential detections without significantly increasing false positives. Data augmentation techniques, such as generating synthetic samples and applying transformations, can enhance the representation of underrepresented landslide features. Additionally, incorporating hybrid attention mechanisms or multi-scale feature extraction strategies may allow the model to capture more complex landslide patterns effectively. Besides, post-processing techniques such as morphological operations or region-growing algorithms could be utilized to refine detections and improve recall.</p>
<p>Additionally, from <xref ref-type="table" rid="table-1">Table 1</xref>, we observe that models integrated with attention mechanisms generally enhance detection performance in terms of precision, recall, and F-score, but at the cost of increased GFLOPs, parameter size, and computational time. For instance, YOLOv5n &#x002B; CBAM improves the F-score and mAP@0.5 compared to the baseline YOLOv5n while maintaining a relatively small increase in computational complexity. This suggests that CBAM enhances feature extraction efficiency with only a small computational overhead. However, a different trend is seen in YOLOv5s &#x002B; ECA, which achieves a higher precision (80.7%) but at a significant computational cost and a GPU memory demand. This highlights that while ECA-based models improve accuracy, they also require substantially more computational resources. Based on the results, we noted that the YOLOv5n &#x002B; CBAM model presents a strong balance between performance and efficiency, making it an ideal choice for applications where both accuracy and computational constraints need to be considered. On the other hand, if higher accuracy is the priority and computational resources are not a constraint, models like YOLOv5s &#x002B; ECA can be beneficial. Furthermore, we discuss the performance of an additional attention mechanism integrated within YOLOv5 to infer its capability. Next, we evaluate the potential of the state-of-the-art YOLO-based models for extracting landslides. Afterward, the implication of the patience parameter in the YOLO model is highlighted.</p>

<sec id="s4_4_1">
<label>4.4.1</label>
<title>Additional Attention Mechanism-Based Inferences</title>
<p>We deliberated on the supplementary attention model to determine the appropriate model for landslide detection. Specifically, GAM (global attention mechanism) [<xref ref-type="bibr" rid="ref-70">70</xref>] restructures the channel and spatial modules of CBAM innovatively. A major improvement (&#x002B;5.4%) is noted in the f-score (another evaluation indicator) of YOLOv5s &#x002B; GAM (75.6%), in comparison to the YOLOv5n (70.2%). The computed mAP@0.50 of YOLOv5n &#x002B; GAM &#x003D; 74.4%, and YOLOv5s &#x002B; GAM &#x003D; 75.9%. Compared with the baseline models (YOLOv5n &#x003D; 73.1%, and YOLOv5s &#x003D; 75.1%), progress of &#x002B;1.3%, and &#x002B;0.8% is found in the case of YOLOv5n &#x002B; GAM, and YOLOv5s &#x002B; GAM, respectively. Additionally, the precision (YOLOv5n &#x002B; GAM &#x003D; 73.8%, and YOLOv5s &#x002B; GAM &#x003D; 74.9%) and recall (YOLOv5s &#x002B; GAM &#x003D; 76.3%) of GAM is improved than YOLOv5n (precision &#x003D; 71.3%, and recall &#x003D; 74.5%), and YOLOv5s (precision &#x003D; 73.3%, and recall &#x003D; 67.4%). However, the computed recall (71.6%) of YOLOv5n &#x002B; GAM is less than the baseline model. Furthermore, there was a notable difference of 15.8 GFLOPs between YOLOv5n &#x002B; GAM (5.5) and YOLOv5s &#x002B; GAM (21.3), indicating increased computational complexity for the latter. The number of parameters in YOLOv5s &#x002B; GAM (21.3 million) was also significantly greater than that of YOLOv5n &#x002B; GAM (5.5 million). Subsequently, the execution time of YOLOv5s &#x002B; GAM (1.46 h) was higher than that of YOLOv5n &#x002B; GAM (1.06 h) network, reflecting the increased resource demands of the more complex architecture.</p>
</sec>
<sec id="s4_4_2">
<label>4.4.2</label>
<title>Importance of Early Stopping Conditions</title>
<p>During the experimentation with machine learning and deep learning, early stopping is a method applied to aid the network&#x2019;s efficacy. It involves the constant monitoring of evaluation indicators while training and terminating the training procedure if no enhancement is observed after certain epochs, technically called the patience parameter (which is user-defined). It helps achieve a balance between training time and overall model performance.</p>
<p>Selecting the appropriate patience parameter is key to improving performance. If the patience is set too low, the model training might end prematurely, potentially overlooking possible performance enhancements. Conversely, an excessively high patience value can needlessly prolong the training duration, resulting in the inefficient use of computing resources and time. Moreover, there is no definite procedure for deciding an optimal value for patience; however, it depends on numerous factors like dataset characteristics, speed, algorithm complexity, and the available computational or hardware resources. Determining the optimal value demands a process of experimentation and fine-tuning. By default, the patience value of the YOLOv5 model is 100, which was therefore used in our experiments. The patience parameter can be enabled or disabled (set patience &#x003D; 0) easily. <xref ref-type="fig" rid="fig-7">Fig. 7</xref> illustrates the result of the patience parameter on the baseline model (YOLOv5n, YOLOv5s, and YOLOv5m) and the combination of baseline and attention models. Amongst the baseline models, YOLOv5n was trained for maximum epochs (324), whereas YOLOv5s was trained for 217 epochs, and YOLOv5m for 195 epochs. The fact that YOLOv5n trained for the maximum number of epochs among the baseline models suggests that it benefited from additional training time, potentially leading to better learning of complex features.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Results of the early stopping condition: baseline and proposed model</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMES_64395-fig-7.tif"/>
</fig>
<p>In the case of attention models, YOLOv5n &#x002B; CBAM was trained for 492 epochs (highest), followed by YOLOv5n &#x002B; ECA (446). However, the training of YOLOv5n &#x002B; CBAM (263), YOLOv5n &#x002B; ECA (286), and YOLOv5s &#x002B; CBAM (272) stopped early. <xref ref-type="fig" rid="fig-7">Fig. 7</xref> also demonstrates the epoch at which the best results of each model were achieved. We aim to emphasize the impact of the &#x2018;patience&#x2019; parameter on training duration and model performance. Notably, the attention models, particularly YOLOv5n &#x002B; CBAM and YOLOv5n &#x002B; ECA, reached higher epoch counts, indicating their ability to learn more effectively with the added complexity of the attention mechanisms. Hence, a higher patience value allows models to explore their learning capacity more fully, which can lead to improved performance. However, the early stopping of some models suggests that not all configurations benefit equally from extended training periods, highlighting the importance of monitoring performance metrics throughout training. Moreover, different datasets may yield varying training dynamics, potentially influencing how effectively the models learn over extended epochs. Therefore, while the results are promising for the datasets used in this study, they may not generalize across all scenarios.</p>

</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusion</title>
<p>This article has presented an innovative attention-enhanced YOLOv5 network for landslide detection from satellite imagery. The novelty lies in the integration of two lightweight though powerful attention models, specifically, CBAM and ECA, into the backbone and neck of the YOLOv5 architecture, respectively. Notably, the YOLOv5n &#x002B; CBAM configuration achieved the highest detection accuracy among the evaluated models, attaining a mAP@0.5 of 77.8%, closely followed by the YOLOv5n &#x002B; ECA variant, which yielded a mAP@0.5 of 77.0%. The YOLOv5n &#x002B; CBAM configuration demonstrated notable improvements in detection accuracy, outperforming existing methods by effectively focusing on key landscape features that are crucial for precise landslide mapping. The incorporation of attention mechanisms significantly enhances the model&#x2019;s ability to differentiate between subtle landscape variations, thereby improving its performance under complex and challenging terrain conditions. While the current findings are promising, they are still preliminary; the model&#x2019;s true potential will be more convincingly demonstrated through comprehensive case studies and large-scale image analysis in future work. We aim to explore additional YOLO variants and attention models to further refine and expand the capabilities of this approach. Eventually, the goal is to develop robust, real-time systems for landslide detection that can be seamlessly integrated into disaster management frameworks, thereby supporting mitigation efforts and enhancing resilience to natural hazards. This study contributes to the field by advancing machine learning applications in environmental monitoring, offering a promising solution to the challenges of real-time landslide mapping.</p>
</sec>
</body>
<back>
<ack>
<p>The authors would like to thank the director of the Wadia Institute of Himalayan Geology, Dehradun, for his motivation and encouragement. This work&#x2019;s contribution number is WIHG/255.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This work is supported by the Department of Science and Technology, Science and Engineering Research Board, New Delhi, India, under Grant No. EEQ/2022/000812. Also funded by the Centre of Advanced Modelling and Geospatial Information Systems (CAMGIS), University of Technology Sydney.</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>Conceptualization, data collection, implementation, validation, manuscript writing, and funding acquisition: Naveen Chandra; implementation, validation, and manuscript editing: Himadri Vaidya; manuscript editing: Suraj Sawant; manuscript review and editing: Shilpa Gite; manuscript editing and critical review: Biswajeet Pradhan; funding for publication: Biswajeet Pradhan. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>The dataset used for this study is freely available at <ext-link ext-link-type="uri" xlink:href="https://github.com/Abbott-max/dataset">https://github.com/Abbott-max/dataset</ext-link> (accessed on 01 January 2025).</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Casagli</surname> <given-names>N</given-names></string-name>, <string-name><surname>Intrieri</surname> <given-names>E</given-names></string-name>, <string-name><surname>Tofani</surname> <given-names>V</given-names></string-name>, <string-name><surname>Gigli</surname> <given-names>G</given-names></string-name>, <string-name><surname>Raspini</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Landslide detection, monitoring and prediction with remote-sensing techniques</article-title>. <source>Nat Rev Earth Environ</source>. <year>2023</year>;<volume>4</volume>(<issue>1</issue>):<fpage>51</fpage>&#x2013;<lpage>64</lpage>. doi:<pub-id pub-id-type="doi">10.1038/s43017-022-00373-x</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Xu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ouyang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>B</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>CAS landslide dataset: a large-scale and multisensor dataset for deep learning-based landslide detection</article-title>. <source>Sci Data</source>. <year>2024</year>;<volume>11</volume>(<issue>1</issue>):<fpage>12</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41597-023-02847-z</pub-id>; <pub-id pub-id-type="pmid">38168493</pub-id></mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ma</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Mei</surname> <given-names>G</given-names></string-name>, <string-name><surname>Piccialli</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Machine learning for landslides prevention: a survey</article-title>. <source>Neural Comput Appl</source>. <year>2021</year>;<volume>33</volume>(<issue>17</issue>):<fpage>10881</fpage>&#x2013;<lpage>907</lpage>. doi:<pub-id pub-id-type="doi">10.36227/techrxiv.12546098.v1</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ma</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Mei</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Deep learning for geological hazards analysis: data, models, applications, and opportunities</article-title>. <source>Earth Sci Rev</source>. <year>2021</year>;<volume>223</volume>:<fpage>103858</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.earscirev.2021.103858</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Han</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Fang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>B</given-names></string-name></person-group>. <article-title>A novel Dynahead-Yolo neural network for the detection of landslides with variable proportions using remote sensing images</article-title>. <source>Front Earth Sci</source>. <year>2023</year>;<volume>10</volume>:<fpage>1077153</fpage>. doi:<pub-id pub-id-type="doi">10.3389/feart.2022.1077153</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>J</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Z</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A research on landslides automatic extraction model based on the improved mask R-CNN</article-title>. <source>ISPRS Int J Geo Inf</source>. <year>2021</year>;<volume>10</volume>(<issue>3</issue>):<fpage>168</fpage>. doi:<pub-id pub-id-type="doi">10.3390/ijgi10030168</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Stumpf</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kerle</surname> <given-names>N</given-names></string-name>, <string-name><surname>Casagli</surname> <given-names>N</given-names></string-name></person-group>. <article-title>Object-oriented change detection for landslide rapid mapping</article-title>. <source>IEEE Geosci Remote Sens Lett</source>. <year>2011</year>;<volume>8</volume>(<issue>4</issue>):<fpage>701</fpage>&#x2013;<lpage>5</lpage>. doi:<pub-id pub-id-type="doi">10.1109/LGRS.2010.2101045</pub-id>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Blaschke</surname> <given-names>T</given-names></string-name>, <string-name><surname>Feizizadeh</surname> <given-names>B</given-names></string-name>, <string-name><surname>H&#x00F6;lbling</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Object-based image analysis and digital terrain analysis for locating landslides in the Urmia Lake basin</article-title>. <source>Iran IEEE J Sel Top Appl Earth Obs Remote Sens</source>. <year>2014</year>;<volume>7</volume>(<issue>12</issue>):<fpage>4806</fpage>&#x2013;<lpage>17</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JSTARS.2014.2350036</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dong</surname> <given-names>Z</given-names></string-name>, <string-name><surname>An</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>D</given-names></string-name></person-group>. <article-title>L-UNet: a landslide extraction model using multi-scale feature fusion and attention mechanism</article-title>. <source>Remote Sens</source>. <year>2022</year>;<volume>14</volume>(<issue>11</issue>):<fpage>2552</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs14112552</pub-id>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Keyport</surname> <given-names>RN</given-names></string-name>, <string-name><surname>Oommen</surname> <given-names>T</given-names></string-name>, <string-name><surname>Martha</surname> <given-names>TR</given-names></string-name>, <string-name><surname>Sajinkumar</surname> <given-names>KS</given-names></string-name>, <string-name><surname>Gierke</surname> <given-names>JS</given-names></string-name></person-group>. <article-title>A comparative analysis of pixel- and object-based detection of landslides from very high-resolution images</article-title>. <source>Int J Appl Earth Obs Geoinf</source>. <year>2018</year>;<volume>64</volume>(<issue>6</issue>):<fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.jag.2017.08.015</pub-id>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ma</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Ranjan</surname> <given-names>R</given-names></string-name>, <string-name><surname>Zomaya</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Remote sensing big data computing: challenges and opportunities</article-title>. <source>Future Gener Comput Syst</source>. <year>2015</year>;<volume>51</volume>(<issue>1</issue>):<fpage>47</fpage>&#x2013;<lpage>60</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.future.2014.10.029</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>YN</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Deep learning for processing and analysis of remote sensing big data: a technical review</article-title>. <source>Big Earth Data</source>. <year>2022</year>;<volume>6</volume>(<issue>4</issue>):<fpage>527</fpage>&#x2013;<lpage>60</lpage>. doi:<pub-id pub-id-type="doi">10.1080/20964471.2021.1964879</pub-id>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Huang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Review on landslide susceptibility mapping using support vector machines</article-title>. <source>Catena</source>. <year>2018</year>;<volume>165</volume>(<issue>1&#x2013;2</issue>):<fpage>520</fpage>&#x2013;<lpage>9</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.catena.2018.03.003</pub-id>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Sharma</surname> <given-names>N</given-names></string-name>, <string-name><surname>Saharia</surname> <given-names>M</given-names></string-name>, <string-name><surname>Ramana</surname> <given-names>GV</given-names></string-name></person-group>. <article-title>High resolution landslide susceptibility mapping using ensemble machine learning and geospatial big data</article-title>. <source>Catena</source>. <year>2024</year>;<volume>235</volume>(<issue>1</issue>):<fpage>107653</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.catena.2023.107653</pub-id>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Yin</surname> <given-names>K</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Landslide identification using machine learning</article-title>. <source>Geosci Front</source>. <year>2021</year>;<volume>12</volume>(<issue>1</issue>):<fpage>351</fpage>&#x2013;<lpage>64</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.gsf.2020.02.012</pub-id>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lombardo</surname> <given-names>L</given-names></string-name>, <string-name><surname>Mai</surname> <given-names>PM</given-names></string-name></person-group>. <article-title>Presenting logistic regression-based landslide susceptibility results</article-title>. <source>Eng Geol</source>. <year>2018</year>;<volume>244</volume>(<issue>1</issue>):<fpage>14</fpage>&#x2013;<lpage>24</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.enggeo.2018.07.019</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>F</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Improving the accuracy of landslide detection in &#x0201C;off-site&#x0201D; area by machine learning model portability comparison: a case study of Jiuzhaigou earthquake, China</article-title>. <source>Remote Sens</source>. <year>2019</year>;<volume>11</volume>(<issue>21</issue>):<fpage>2530</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs11212530</pub-id>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Fang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Duan</surname> <given-names>G</given-names></string-name>, <string-name><surname>Peng</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Landslide susceptibility mapping using rotation forest ensemble technique with different decision trees in the Three Gorges Reservoir area, China</article-title>. <source>Remote Sens</source>. <year>2021</year>;<volume>13</volume>(<issue>2</issue>):<fpage>238</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs13020238</pub-id>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Minaee</surname> <given-names>S</given-names></string-name>, <string-name><surname>Boykov</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Porikli</surname> <given-names>F</given-names></string-name>, <string-name><surname>Plaza</surname> <given-names>A</given-names></string-name>, <string-name><surname>Kehtarnavaz</surname> <given-names>N</given-names></string-name>, <string-name><surname>Terzopoulos</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Image segmentation using deep learning: a survey</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. <year>2022</year>;<volume>44</volume>(<issue>7</issue>):<fpage>3523</fpage>&#x2013;<lpage>42</lpage>. doi:<pub-id pub-id-type="doi">10.1109/tpami.2021.3059968</pub-id>; <pub-id pub-id-type="pmid">33596172</pub-id></mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhao</surname> <given-names>ZQ</given-names></string-name>, <string-name><surname>Zheng</surname> <given-names>P</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>ST</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Object detection with deep learning: a review</article-title>. arXiv:1807.05511v2. <year>2018</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zeng</surname> <given-names>D</given-names></string-name>, <string-name><surname>Liao</surname> <given-names>M</given-names></string-name>, <string-name><surname>Tavakolian</surname> <given-names>M</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>B</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>D</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Deep learning for scene classification: a survey</article-title>. <comment>arXiv:2101.10531. 2021</comment>. </mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Jia</surname> <given-names>J</given-names></string-name>, <string-name><surname>Ye</surname> <given-names>W</given-names></string-name></person-group>. <article-title>Deep learning for earthquake disaster assessment: objects, data, models, stages, challenges, and opportunities</article-title>. <source>Remote Sens</source>. <year>2023</year>;<volume>15</volume>(<issue>16</issue>):<fpage>4098</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs15164098</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bentivoglio</surname> <given-names>R</given-names></string-name>, <string-name><surname>Isufi</surname> <given-names>E</given-names></string-name>, <string-name><surname>Jonkman</surname> <given-names>SN</given-names></string-name>, <string-name><surname>Taormina</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Deep learning methods for flood mapping: a review of existing applications and future research directions</article-title>. <source>Hydrol Earth Syst Sci</source>. <year>2022</year>;<volume>26</volume>(<issue>16</issue>):<fpage>4345</fpage>&#x2013;<lpage>78</lpage>. doi:<pub-id pub-id-type="doi">10.5194/hess-26-4345-2022</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mohan</surname> <given-names>A</given-names></string-name>, <string-name><surname>Singh</surname> <given-names>AK</given-names></string-name>, <string-name><surname>Kumar</surname> <given-names>B</given-names></string-name>, <string-name><surname>Dwivedi</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Review on remote sensing methods for landslide detection using machine and deep learning</article-title>. <source>Trans Emerging Tel Tech</source>. <year>2021</year>;<volume>32</volume>(<issue>7</issue>):<fpage>e3998</fpage>. doi:<pub-id pub-id-type="doi">10.1002/ett.3998</pub-id>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cheng</surname> <given-names>G</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>C</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Advances in deep learning recognition of landslides based on remote sensing images</article-title>. <source>Remote Sens</source>. <year>2024</year>;<volume>16</volume>(<issue>10</issue>):<fpage>1787</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs16101787</pub-id>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Shi</surname> <given-names>W</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>M</given-names></string-name>, <string-name><surname>Ke</surname> <given-names>H</given-names></string-name>, <string-name><surname>Fang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhan</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Landslide recognition by deep convolutional neural network and change detection</article-title>. <source>IEEE Trans Geosci Remote Sens</source>. <year>2021</year>;<volume>59</volume>(<issue>6</issue>):<fpage>4654</fpage>&#x2013;<lpage>72</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TGRS.2020.3015826</pub-id>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Tehrani</surname> <given-names>FS</given-names></string-name>, <string-name><surname>Calvello</surname> <given-names>M</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Lacasse</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Machine learning and landslide studies: recent advances and applications</article-title>. <source>Nat Hazards</source>. <year>2022</year>;<volume>114</volume>(<issue>2</issue>):<fpage>1197</fpage>&#x2013;<lpage>245</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11069-022-05423-7</pub-id>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Akosah</surname> <given-names>S</given-names></string-name>, <string-name><surname>Gratchev</surname> <given-names>I</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>DH</given-names></string-name>, <string-name><surname>Ohn</surname> <given-names>SY</given-names></string-name></person-group>. <article-title>Application of artificial intelligence and remote sensing for landslide detection and prediction: systematic review</article-title>. <source>Remote Sens</source>. <year>2024</year>;<volume>16</volume>(<issue>16</issue>):<fpage>2947</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs16162947</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bragagnolo</surname> <given-names>L</given-names></string-name>, <string-name><surname>Rezende</surname> <given-names>LR</given-names></string-name>, <string-name><surname>da Silva</surname> <given-names>RV</given-names></string-name>, <string-name><surname>Grzybowski</surname> <given-names>JMV</given-names></string-name></person-group>. <article-title>Convolutional neural networks applied to semantic segmentation of landslide scars</article-title>. <source>Catena</source>. <year>2021</year>;<volume>201</volume>(<issue>1</issue>):<fpage>105189</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.catena.2021.105189</pub-id>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ghorbanzadeh</surname> <given-names>O</given-names></string-name>, <string-name><surname>Crivellari</surname> <given-names>A</given-names></string-name>, <string-name><surname>Ghamisi</surname> <given-names>P</given-names></string-name>, <string-name><surname>Shahabi</surname> <given-names>H</given-names></string-name>, <string-name><surname>Blaschke</surname> <given-names>T</given-names></string-name></person-group>. <article-title>A comprehensive transferability evaluation of U-Net and ResU-Net for landslide detection from Sentinel-2 data (case study areas from Taiwan, China, and Japan)</article-title>. <source>Sci Rep</source>. <year>2021</year>;<volume>11</volume>(<issue>1</issue>):<fpage>14629</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-021-94190-9</pub-id>; <pub-id pub-id-type="pmid">34272463</pub-id></mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Ghorbanzadeh</surname> <given-names>O</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ghamisi</surname> <given-names>P</given-names></string-name>, <string-name><surname>Kopp</surname> <given-names>M</given-names></string-name>, <string-name><surname>Kreil</surname> <given-names>D</given-names></string-name></person-group>. <article-title>Landslide4Sense: reference benchmark data and deep learning models for landslide detection</article-title>. <comment>arXiv:2206.00515. 2022</comment>. </mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Meena</surname> <given-names>SR</given-names></string-name>, <string-name><surname>Nava</surname> <given-names>L</given-names></string-name>, <string-name><surname>Bhuyan</surname> <given-names>K</given-names></string-name>, <string-name><surname>Puliero</surname> <given-names>S</given-names></string-name>, <string-name><surname>Soares</surname> <given-names>LP</given-names></string-name>, <string-name><surname>Dias</surname> <given-names>HC</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>HR-GLDD: a globally distributed dataset using generalized DL for rapid landslide mapping on HR satellite imagery</article-title>. <source>Earth Syst Sci Data</source>. <year>2023</year>;<volume>15</volume>(<issue>7</issue>):<fpage>3283</fpage>&#x2013;<lpage>98</lpage>. doi:<pub-id pub-id-type="doi">10.5194/essd-2022-350</pub-id>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Devara</surname> <given-names>M</given-names></string-name>, <string-name><surname>Maurya</surname> <given-names>VK</given-names></string-name>, <string-name><surname>Dwivedi</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Landslide extraction using a novel empirical method and binary semantic segmentation U-NET framework using sentinel-2 imagery</article-title>. <source>Remote Sens Lett</source>. <year>2024</year>;<volume>15</volume>(<issue>3</issue>):<fpage>326</fpage>&#x2013;<lpage>38</lpage>. doi:<pub-id pub-id-type="doi">10.1080/2150704x.2024.2320178</pub-id>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>T</given-names></string-name>, <string-name><surname>Niu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Plaza</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Landslide detection mapping employing CNN, ResNet, and DenseNet in the Three Gorges Reservoir, China</article-title>. <source>IEEE J Sel Top Appl Earth Obs Remote Sens</source>. <year>2021</year>;<volume>14</volume>:<fpage>11417</fpage>&#x2013;<lpage>28</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JSTARS.2021.3117975</pub-id>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ullo</surname> <given-names>SL</given-names></string-name>, <string-name><surname>Mohan</surname> <given-names>A</given-names></string-name>, <string-name><surname>Sebastianelli</surname> <given-names>A</given-names></string-name>, <string-name><surname>Ahamed</surname> <given-names>SE</given-names></string-name>, <string-name><surname>Kumar</surname> <given-names>B</given-names></string-name>, <string-name><surname>Dwivedi</surname> <given-names>R</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A new mask R-CNN-based method for improved landslide detection</article-title>. <source>IEEE J Sel Top Appl Earth Obs Remote Sens</source>. <year>2021</year>;<volume>14</volume>:<fpage>3799</fpage>&#x2013;<lpage>810</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JSTARS.2021.3064981</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Fu</surname> <given-names>R</given-names></string-name>, <string-name><surname>He</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>G</given-names></string-name>, <string-name><surname>Li</surname> <given-names>W</given-names></string-name>, <string-name><surname>Mao</surname> <given-names>J</given-names></string-name>, <string-name><surname>He</surname> <given-names>M</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Fast seismic landslide detection based on improved mask R-CNN</article-title>. <source>Remote Sens</source>. <year>2022</year>;<volume>14</volume>(<issue>16</issue>):<fpage>3928</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs14163928</pub-id>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Landslide detection with Mask R-CNN using complex background enhancement based on multi-scale samples</article-title>. <source>Geomat Nat Hazards Risk</source>. <year>2024</year>;<volume>15</volume>(<issue>1</issue>):<fpage>2300823</fpage>. doi:<pub-id pub-id-type="doi">10.1080/19475705.2023.2300823</pub-id>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Diwan</surname> <given-names>T</given-names></string-name>, <string-name><surname>Anirudh</surname> <given-names>G</given-names></string-name>, <string-name><surname>Tembhurne</surname> <given-names>JV</given-names></string-name></person-group>. <article-title>Object detection using YOLO: challenges, architectural successors, datasets and applications</article-title>. <source>Multimed Tools Appl</source>. <year>2023</year>;<volume>82</volume>(<issue>6</issue>):<fpage>9243</fpage>&#x2013;<lpage>75</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11042-022-13644-y</pub-id>; <pub-id pub-id-type="pmid">35968414</pub-id></mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ju</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Jin</surname> <given-names>S</given-names></string-name>, <string-name><surname>Li</surname> <given-names>W</given-names></string-name>, <string-name><surname>Su</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Dong</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Loess landslide detection using object detection algorithms in northwest China</article-title>. <source>Remote Sens</source>. <year>2022</year>;<volume>14</volume>(<issue>5</issue>):<fpage>1182</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs14051182</pub-id>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Pang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>G</given-names></string-name>, <string-name><surname>He</surname> <given-names>J</given-names></string-name>, <string-name><surname>Li</surname> <given-names>W</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Automatic remote sensing identification of co-seismic landslides using deep learning methods</article-title>. <source>Forests</source>. <year>2022</year>;<volume>13</volume>(<issue>8</issue>):<fpage>1213</fpage>. doi:<pub-id pub-id-type="doi">10.3390/f13081213</pub-id>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Methods for landslide detection based on lightweight YOLOv4 convolutional neural network</article-title>. <source>Earth Sci Inform</source>. <year>2022</year>;<volume>15</volume>(<issue>2</issue>):<fpage>765</fpage>&#x2013;<lpage>75</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s12145-022-00764-0</pub-id>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mo</surname> <given-names>P</given-names></string-name>, <string-name><surname>Li</surname> <given-names>D</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Jia</surname> <given-names>J</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>X</given-names></string-name></person-group>. <article-title>A lightweight and partitioned CNN algorithm for multi-landslide detection in remote sensing images</article-title>. <source>Appl Sci</source>. <year>2023</year>;<volume>13</volume>(<issue>15</issue>):<fpage>8583</fpage>. doi:<pub-id pub-id-type="doi">10.3390/app13158583</pub-id>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chandra</surname> <given-names>N</given-names></string-name>, <string-name><surname>Vaidya</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Automated detection of landslide events from multi-source remote sensing imagery: performance evaluation and analysis of YOLO algorithms</article-title>. <source>J Earth Syst Sci</source>. <year>2024</year>;<volume>133</volume>(<issue>3</issue>):<fpage>127</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s12040-024-02327-x</pub-id>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>SE-YOLOv7 landslide detection algorithm based on attention mechanism and improved loss function</article-title>. <source>Land</source>. <year>2023</year>;<volume>12</volume>(<issue>8</issue>):<fpage>1522</fpage>. doi:<pub-id pub-id-type="doi">10.3390/land12081522</pub-id>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Niu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Li</surname> <given-names>B</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Potential landslide identification based on improved YOLOv8 and InSAR phase-gradient stacking</article-title>. <source>IEEE J Sel Top Appl Earth Obs Remote Sens</source>. <year>2024</year>;<volume>17</volume>(<issue>5</issue>):<fpage>10367</fpage>&#x2013;<lpage>76</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JSTARS.2024.3399788</pub-id>.</mixed-citation></ref>
<ref id="ref-46"><label>[46]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hou</surname> <given-names>H</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>M</given-names></string-name>, <string-name><surname>Tie</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>W</given-names></string-name></person-group>. <article-title>A universal landslide detection method in optical remote sensing images based on improved YOLOX</article-title>. <source>Remote Sens</source>. <year>2022</year>;<volume>14</volume>(<issue>19</issue>):<fpage>4939</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs14194939</pub-id>.</mixed-citation></ref>
<ref id="ref-47"><label>[47]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ji</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>D</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>C</given-names></string-name>, <string-name><surname>Li</surname> <given-names>W</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Q</given-names></string-name></person-group>. <article-title>Landslide detection from an open satellite imagery and digital elevation model dataset using attention boosted convolutional neural networks</article-title>. <source>Landslides</source>. <year>2020</year>;<volume>17</volume>(<issue>6</issue>):<fpage>1337</fpage>&#x2013;<lpage>52</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10346-020-01353-2</pub-id>.</mixed-citation></ref>
<ref id="ref-48"><label>[48]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zeng</surname> <given-names>S</given-names></string-name>, <string-name><surname>Xiao</surname> <given-names>K</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Yao</surname> <given-names>G</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A novel landslide identification method for multi-scale and complex background region based on multi-model fusion: YOLO &#x002B; U-Net</article-title>. <source>Landslides</source>. <year>2024</year>;<volume>21</volume>(<issue>4</issue>):<fpage>901</fpage>&#x2013;<lpage>17</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10346-023-02184-7</pub-id>.</mixed-citation></ref>
<ref id="ref-49"><label>[49]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Miao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Lightweight attention-guided YOLO with level set layer for landslide detection from optical satellite images</article-title>. <source>IEEE J Sel Top Appl Earth Obs Remote Sens</source>. <year>2024</year>;<volume>17</volume>:<fpage>3543</fpage>&#x2013;<lpage>59</lpage>. doi:<pub-id pub-id-type="doi">10.1109/JSTARS.2024.3351277</pub-id>.</mixed-citation></ref>
<ref id="ref-50"><label>[50]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>TT</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>YH</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>ZH</given-names></string-name></person-group>. <article-title>Intelligent identification of landslides in loess areas based on the improved YOLO algorithm: a case study of loess landslides in Baoji City</article-title>. <source>J Mt Sci</source>. <year>2023</year>;<volume>20</volume>(<issue>11</issue>):<fpage>3343</fpage>&#x2013;<lpage>59</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11629-023-8128-0</pub-id>.</mixed-citation></ref>
<ref id="ref-51"><label>[51]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cheng</surname> <given-names>L</given-names></string-name>, <string-name><surname>Li</surname> <given-names>J</given-names></string-name>, <string-name><surname>Duan</surname> <given-names>P</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>M</given-names></string-name></person-group>. <article-title>A small attentional YOLO model for landslide detection from satellite remote sensing images</article-title>. <source>Landslides</source>. <year>2021</year>;<volume>18</volume>(<issue>8</issue>):<fpage>2751</fpage>&#x2013;<lpage>65</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s10346-021-01694-6</pub-id>.</mixed-citation></ref>
<ref id="ref-52"><label>[52]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Jia</surname> <given-names>L</given-names></string-name>, <string-name><surname>Leng</surname> <given-names>X</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Nie</surname> <given-names>M</given-names></string-name></person-group>. <article-title>Recognizing landslides in remote sensing images based on enhancement of information in digital elevation models</article-title>. <source>Remote Sens Lett</source>. <year>2024</year>;<volume>15</volume>(<issue>3</issue>):<fpage>224</fpage>&#x2013;<lpage>32</lpage>. doi:<pub-id pub-id-type="doi">10.1080/2150704x.2024.2313611</pub-id>.</mixed-citation></ref>
<ref id="ref-53"><label>[53]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Song</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>G</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>F</given-names></string-name>, <string-name><surname>Li</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Automatic recognition of landslides based on YOLOv7 and attention mechanism</article-title>. <source>J Mt Sci</source>. <year>2024</year>;<volume>21</volume>(<issue>8</issue>):<fpage>2681</fpage>&#x2013;<lpage>95</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s11629-024-8669-x</pub-id>.</mixed-citation></ref>
<ref id="ref-54"><label>[54]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>S</given-names></string-name>, <string-name><surname>Qi</surname> <given-names>W</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>T</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>LS-YOLO: a novel model for detecting multiscale landslides with remote sensing images</article-title>. <source>IEEE J Sel Top Appl Earth Obs Remote Sens</source>. <year>2024</year>;<volume>17</volume>:<fpage>4952</fpage>&#x2013;<lpage>65</lpage>. doi:<pub-id pub-id-type="doi">10.1109/jstars.2024.3363160</pub-id>.</mixed-citation></ref>
<ref id="ref-55"><label>[55]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Lei</surname> <given-names>H</given-names></string-name>, <string-name><surname>Jian</surname> <given-names>W</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>N</given-names></string-name></person-group>. <article-title>Enhancing landslide detection: a novel LA-YOLO model for rainfall-induced shallow landslides</article-title>. <source>IEEE Geosci Remote Sens Lett</source>. <year>2025</year>;<volume>22</volume>:<fpage>6004905</fpage>. doi:<pub-id pub-id-type="doi">10.1109/LGRS.2025.3541867</pub-id>.</mixed-citation></ref>
<ref id="ref-56"><label>[56]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Redmon</surname> <given-names>J</given-names></string-name>, <string-name><surname>Divvala</surname> <given-names>S</given-names></string-name>, <string-name><surname>Girshick</surname> <given-names>R</given-names></string-name>, <string-name><surname>Farhadi</surname> <given-names>A</given-names></string-name></person-group>. <article-title>You only look once: unified, real-time object detection</article-title>. In: <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR); 2016 Jun 27&#x2013;30; Las Vegas, NV, USA</conf-name>. p. <fpage>779</fpage>&#x2013;<lpage>88</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2016.91</pub-id>.</mixed-citation></ref>
<ref id="ref-57"><label>[57]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Terven</surname> <given-names>J</given-names></string-name>, <string-name><surname>C&#x00F3;rdova-Esparza</surname> <given-names>DM</given-names></string-name>, <string-name><surname>Romero-Gonz&#x00E1;lez</surname> <given-names>JA</given-names></string-name></person-group>. <article-title>A comprehensive review of YOLO architectures in computer vision: from YOLOv1 to YOLOv8 and YOLO-NAS</article-title>. <source>Mach Learn Knowl Extr</source>. <year>2023</year>;<volume>5</volume>(<issue>4</issue>):<fpage>1680</fpage>&#x2013;<lpage>716</lpage>. doi:<pub-id pub-id-type="doi">10.3390/make5040083</pub-id>.</mixed-citation></ref>
<ref id="ref-58"><label>[58]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>CY</given-names></string-name>, <string-name><surname>Mark Liao</surname> <given-names>HY</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>YH</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>PY</given-names></string-name>, <string-name><surname>Hsieh</surname> <given-names>JW</given-names></string-name>, <string-name><surname>Yeh</surname> <given-names>IH</given-names></string-name></person-group>. <article-title>CSPNet: a new backbone that can enhance learning capability of CNN</article-title>. In: <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW); 2020 Jun 14&#x2013;19; Seattle, WA, USA</conf-name>. p. <fpage>1571</fpage>&#x2013;<lpage>80</lpage>. doi:<pub-id pub-id-type="doi">10.1109/cvprw50498.2020.00203</pub-id>.</mixed-citation></ref>
<ref id="ref-59"><label>[59]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>F</given-names></string-name>, <string-name><surname>Gu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>L</given-names></string-name></person-group>. <article-title>SF-YOLOv5: a lightweight small object detection algorithm based on improved feature fusion mode</article-title>. <source>Sensors</source>. <year>2022</year>;<volume>22</volume>(<issue>15</issue>):<fpage>5817</fpage>. doi:<pub-id pub-id-type="doi">10.3390/s22155817</pub-id>; <pub-id pub-id-type="pmid">35957375</pub-id></mixed-citation></ref>
<ref id="ref-60"><label>[60]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Qi</surname> <given-names>L</given-names></string-name>, <string-name><surname>Qin</surname> <given-names>H</given-names></string-name>, <string-name><surname>Shi</surname> <given-names>J</given-names></string-name>, <string-name><surname>Jia</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Path aggregation network for instance segmentation</article-title>. In: <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition; 2018 Jun 18&#x2013;23; Salt Lake City, UT, USA</conf-name>. p. <fpage>8759</fpage>&#x2013;<lpage>68</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2018.00913</pub-id>.</mixed-citation></ref>
<ref id="ref-61"><label>[61]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Lin</surname> <given-names>TY</given-names></string-name>, <string-name><surname>Doll&#x00E1;r</surname> <given-names>P</given-names></string-name>, <string-name><surname>Girshick</surname> <given-names>R</given-names></string-name>, <string-name><surname>He</surname> <given-names>K</given-names></string-name>, <string-name><surname>Hariharan</surname> <given-names>B</given-names></string-name>, <string-name><surname>Belongie</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Feature pyramid networks for object detection</article-title>. In: <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR); 2017 Jul 21&#x2013;26; Honolulu, HI, USA</conf-name>. p. <fpage>936</fpage>&#x2013;<lpage>44</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2017.106</pub-id>.</mixed-citation></ref>
<ref id="ref-62"><label>[62]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mumuni</surname> <given-names>A</given-names></string-name>, <string-name><surname>Mumuni</surname> <given-names>F</given-names></string-name></person-group>. <article-title>Data augmentation: a comprehensive survey of modern approaches</article-title>. <source>Array</source>. <year>2022</year>;<volume>16</volume>(<issue>6</issue>):<fpage>100258</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.array.2022.100258</pub-id>.</mixed-citation></ref>
<ref id="ref-63"><label>[63]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Song</surname> <given-names>J</given-names></string-name></person-group>. <article-title>ICIoU: improved loss based on complete intersection over union for bounding box regression</article-title>. <source>IEEE Access</source>. <year>2021</year>;<volume>9</volume>:<fpage>105686</fpage>&#x2013;<lpage>95</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2021.3100414</pub-id>.</mixed-citation></ref>
<ref id="ref-64"><label>[64]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Woo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Park</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>JY</given-names></string-name>, <string-name><surname>Kweon</surname> <given-names>IS</given-names></string-name></person-group>. <article-title>CBAM: convolutional block attention module</article-title>. <comment>arXiv:1807.06521. 2018</comment>.</mixed-citation></ref>
<ref id="ref-65"><label>[65]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Li</surname> <given-names>P</given-names></string-name>, <string-name><surname>Zuo</surname> <given-names>W</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>Q</given-names></string-name></person-group>. <article-title>ECA-net: efficient channel attention for deep convolutional neural networks</article-title>. <comment>arXiv:1910.03151. 2019</comment>.</mixed-citation></ref>
<ref id="ref-66"><label>[66]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Squeeze-and-excitation networks</article-title>. In: <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition; 2018 Jun 18&#x2013;23; Salt Lake City, UT, USA</conf-name>. p. <fpage>7132</fpage>&#x2013;<lpage>41</lpage>.</mixed-citation></ref>
<ref id="ref-67"><label>[67]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhao</surname> <given-names>L</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>C</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Abbas</surname> <given-names>Z</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>YOLOv8-QR: an improved YOLOv8 model via attention mechanism for object detection of QR code defects</article-title>. <source>Comput Electr Eng</source>. <year>2024</year>;<volume>118</volume>(<issue>3</issue>):<fpage>109376</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.compeleceng.2024.109376</pub-id>.</mixed-citation></ref>
<ref id="ref-68"><label>[68]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liang</surname> <given-names>F</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>L</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>To</surname> <given-names>S</given-names></string-name>, <string-name><surname>Abbas</surname> <given-names>Z</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>LAD-Net: a lightweight welding defect surface non-destructive detection algorithm based on the attention mechanism</article-title>. <source>Comput Ind</source>. <year>2024</year>;<volume>161</volume>(<issue>7</issue>):<fpage>104109</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.compind.2024.104109</pub-id>.</mixed-citation></ref>
<ref id="ref-69"><label>[69]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>T</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Landslide detection based on improved YOLOv5 and satellite images</article-title>. In: <conf-name>2021 4th International Conference on Pattern Recognition and Artificial Intelligence (PRAI); 2021 Aug 20&#x2013;22; Yibin, China</conf-name>. p. <fpage>367</fpage>&#x2013;<lpage>71</lpage>. doi:<pub-id pub-id-type="doi">10.1109/prai53619.2021.9551067</pub-id>.</mixed-citation></ref>
<ref id="ref-70"><label>[70]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Shao</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Hoffmann</surname> <given-names>N</given-names></string-name></person-group>. <article-title>Global attention mechanism: retain information to enhance channel-spatial interactions</article-title>. <comment>arXiv:2112.05561. 2021</comment>.</mixed-citation></ref>
</ref-list>
</back></article>