<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">66188</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2025.066188</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>MGD-YOLO: An Enhanced Road Defect Detection Algorithm Based on Multi-Scale Attention Feature Fusion</article-title>
<alt-title alt-title-type="left-running-head">MGD-YOLO: An Enhanced Road Defect Detection Algorithm Based on Multi-Scale Attention Feature Fusion</alt-title>
<alt-title alt-title-type="right-running-head">MGD-YOLO: An Enhanced Road Defect Detection Algorithm Based on Multi-Scale Attention Feature Fusion</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Li</surname><given-names>Zhengji</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Xiong</surname><given-names>Fazhan</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Huang</surname><given-names>Boyun</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Li</surname><given-names>Meihui</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Xiao</surname><given-names>Xi</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Ji</surname><given-names>Yingrui</given-names></name><xref ref-type="aff" rid="aff-3">3</xref><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Xie</surname><given-names>Jiacheng</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>Liang</surname><given-names>Aokun</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-9" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Xu</surname><given-names>Hao</given-names></name><xref ref-type="aff" rid="aff-6">6</xref><email>haxu@bwh.harvard.edu</email></contrib>
<aff id="aff-1"><label>1</label><institution>School of Computer and Software, Chengdu Jincheng College</institution>, <addr-line>Chengdu, 611731</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>College of Arts and Sciences, University of Alabama at Birmingham</institution>, <addr-line>Birmingham</addr-line>, <country>AL 35294</country>, <country>USA</country></aff>
<aff id="aff-3"><label>3</label><institution>Aerospace Information Research Institute</institution>, Chinese Academy of Sciences, <addr-line>Beijing, 100193</addr-line>, <country>China</country></aff>
<aff id="aff-4"><label>4</label><institution>School of Electronic, Electrical and Communication Engineering, University of Chinese Academy of Sciences</institution>, <addr-line>Beijing, 100193</addr-line>, <country>China</country></aff>
<aff id="aff-5"><label>5</label><institution>School of Remote Sensing and Information Engineering, Wuhan University</institution>, <addr-line>Wuhan, 430079</addr-line>, <country>China</country></aff>
<aff id="aff-6"><label>6</label><institution>Department of Medicine, Harvard Medical School</institution>, <addr-line>Boston</addr-line>, <addr-line>MA 02115</addr-line>, <country>USA</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Hao Xu. Email: <email>haxu@bwh.harvard.edu</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2025</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>30</day><month>07</month><year>2025</year>
</pub-date>
<volume>84</volume>
<issue>3</issue>
<fpage>5613</fpage>
<lpage>5635</lpage>
<history>
<date date-type="received">
<day>01</day>
<month>4</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>6</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2025 The Authors.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_66188.pdf"></self-uri>
<abstract>
<p>Accurate and real-time road defect detection is essential for ensuring traffic safety and infrastructure maintenance. However, existing vision-based methods often struggle with small, sparse, and low-resolution defects under complex road conditions. To address these limitations, we propose Multi-Scale Guided Detection YOLO (MGD-YOLO), a novel lightweight and high-performance object detector built upon You Only Look Once Version 5 (YOLOv5). The proposed model integrates three key components: (1) a Multi-Scale Dilated Attention (MSDA) module to enhance semantic feature extraction across varying receptive fields; (2) Depthwise Separable Convolution (DSC) to reduce computational cost and improve model generalization; and (3) a Visual Global Attention Upsampling (VGAU) module that leverages high-level contextual information to refine low-level features for precise localization. Extensive experiments on three public road defect benchmarks demonstrate that MGD-YOLO outperforms state-of-the-art models in both detection accuracy and efficiency. Notably, our model achieves 87.9% accuracy in crack detection, 88.3% overall precision on TD-RD dataset, while maintaining fast inference speed and a compact architecture. These results highlight the potential of MGD-YOLO for deployment in real-time, resource-constrained scenarios, paving the way for practical and scalable intelligent road maintenance systems.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>YOLO</kwd>
<kwd>road damage detection</kwd>
<kwd>object detection</kwd>
<kwd>computer vision</kwd>
<kwd>deep learning</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>Chengdu Jincheng</funding-source>
<award-id>JG2024-1199</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Timely and accurate detection of road surface defects is essential for ensuring transportation safety and enabling proactive infrastructure maintenance. Traditional inspection methods, which rely heavily on manual labor, are often inefficient, costly, and susceptible to human error. With the rapid advancement of deep learning and computer vision technologies, automated visual defect detection has become a promising alternative, particularly for enabling real-time and high-precision assessment of road conditions.</p>
<p>Among existing approaches, one-stage detectors such as the YOLO series have gained popularity due to their efficiency and suitability for real-time applications. However, road defect detection in unconstrained environments presents persistent challenges. As shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>, small irregularly shaped defects often appear under complex lighting or background conditions and are difficult to localize due to their limited semantic features and low contrast. Moreover, conventional convolutional neural networks (CNNs) exhibit limited receptive fields and struggle to model global contextual dependencies, resulting in missed or inaccurate detections for fine-grained or small-scale targets.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Images of some road defects in our dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-1.tif"/>
</fig>
<p>To address these issues, we propose Multi-Scale Guided Detection YOLO (MGD-YOLO), an enhanced YOLOv5-based architecture specifically designed for robust road defect detection. Unlike previous YOLO-based enhancements that primarily focus on speed or general object detection, MGD-YOLO explicitly targets the accurate identification of small-scale, low-contrast defects under complex real-world conditions. It introduces a set of architectural improvements to strengthen the model&#x2019;s ability to capture multi-scale and context-aware features while maintaining a lightweight structure suitable for real-time applications. Specifically, we integrate a Multi-Scale Dilated Attention (MSDA) module into the backbone to capture multi-level contextual dependencies across different receptive fields. We further adopt Depthwise Separable Convolution (DSC) to reduce parameter overhead and computational cost without compromising feature expressiveness. Finally, we design a Visual Global Attention Upsampling (VGAU) module to fuse low-level and high-level features using global semantic guidance, thereby improving the localization and classification of small or low-contrast defects.</p>
<p>Extensive experiments on three public road defect datasets demonstrate that MGD-YOLO achieves superior performance in terms of detection accuracy, robustness, and inference speed compared to existing state-of-the-art methods. Notably, our model achieves a crack detection accuracy of 97.7%, an m<inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:msub><mml:mtext>AP</mml:mtext><mml:mrow><mml:mn>50</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> of 85.7%, and an inference speed of 105 FPS, validating its effectiveness in both accuracy and efficiency for real-time deployment.</p>
<p>Our contributions are summarized as follows:
<list list-type="bullet">
<list-item>
<p>We propose MGD-YOLO, an enhanced YOLOv5-based detector tailored for road defect detection, with particular emphasis on handling small-scale, low-contrast defects in complex environments.</p></list-item>
<list-item>
<p>We introduce a Multi-Scale Dilated Attention (MSDA) module and a Visual Global Attention Upsampling (VGAU) module to improve multi-scale feature representation and enhance semantic consistency across different resolution levels.</p></list-item>
<list-item>
<p>We demonstrate through extensive experiments that MGD-YOLO significantly outperforms existing detectors in both detection accuracy and inference speed, making it highly suitable for deployment in real-time, resource-constrained scenarios.</p></list-item>
</list></p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Work</title>
<sec id="s2_1">
<label>2.1</label>
<title>Deep Learning for Road Defect Detection</title>
<p>Automated road defect detection has attracted growing attention due to its importance for intelligent transportation and infrastructure maintenance. Traditional techniques, such as edge detection [<xref ref-type="bibr" rid="ref-1">1</xref>], wavelet-based analysis [<xref ref-type="bibr" rid="ref-2">2</xref>], and texture descriptors [<xref ref-type="bibr" rid="ref-3">3</xref>], are highly sensitive to noise, lighting variation, and road texture diversity. In contrast, deep learning-based methods [<xref ref-type="bibr" rid="ref-4">4</xref>,<xref ref-type="bibr" rid="ref-5">5</xref>] offer superior robustness and accuracy. Zhang et al. [<xref ref-type="bibr" rid="ref-6">6</xref>] developed a CNN-based pipeline for crack detection that significantly outperformed handcrafted approaches. Shi et al. [<xref ref-type="bibr" rid="ref-7">7</xref>] proposed a structure forest model to handle the complexity and topological variation of cracks. More recently, Park et al. [<xref ref-type="bibr" rid="ref-8">8</xref>] introduced an adaptive pixel neighborhood segmentation method, which improved detection under noisy backgrounds.</p>
<p>YOLO-based detectors have emerged as a strong baseline for real-time road defect detection. For instance, Jocher et al. [<xref ref-type="bibr" rid="ref-9">9</xref>] released YOLOv5, which offers a good balance between speed and performance. Several studies [<xref ref-type="bibr" rid="ref-10">10</xref>&#x2013;<xref ref-type="bibr" rid="ref-14">14</xref>] have adapted YOLO variants to road scenarios, incorporating tailored preprocessing or architectural changes to handle challenges such as occlusion, low resolution, and class imbalance. However, small-scale and unevenly distributed defects (e.g., micro-cracks or edge disintegration) remain under-detected due to limited feature expressiveness in conventional backbones.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Attention Mechanisms in Visual Recognition</title>
<p>Attention mechanisms have become integral in enhancing CNNs&#x2019; and transformers&#x2019; ability to model long-range dependencies and focus on task-relevant features. Channel attention modules, such as SE-Net [<xref ref-type="bibr" rid="ref-15">15</xref>], ECA-Net [<xref ref-type="bibr" rid="ref-16">16</xref>], and the Coordinated Attention (CA) module [<xref ref-type="bibr" rid="ref-17">17</xref>], improve feature channel calibration, enhancing performance across classification and detection tasks. Spatial attention, as used in CBAM [<xref ref-type="bibr" rid="ref-18">18</xref>] and SAM [<xref ref-type="bibr" rid="ref-19">19</xref>], enables focus on key spatial regions, which is particularly useful in defect localization. The combination of spatial and channel attention has also been extended into multi-branch fusion networks and deformable attention [<xref ref-type="bibr" rid="ref-20">20</xref>]. As shown in <xref ref-type="table" rid="table-1">Table 1</xref>, MGD-YOLO achieves the highest mAP with a lightweight architecture compared to recent methods.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Comparison with mainstream detectors used in experiments</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="center">Model</th>
<th align="center">Backbone</th>
<th align="center">Enhancement modules</th>
<th align="center">Benchmark</th>
<th align="center">mAP (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td>YOLOv5s</td>
<td>CSPDarkNet</td>
<td>PANet, SPPF</td>
<td>TD-RD</td>
<td>84.1</td>
</tr>
<tr>
<td>YOLOv6n [<xref ref-type="bibr" rid="ref-21">21</xref>]</td>
<td>EfficientRep</td>
<td>RepOptimizer, Strong augmentations</td>
<td>TD-RD</td>
<td>85.0</td>
</tr>
<tr>
<td>YOLOv7-tiny [<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
<td>E-ELAN</td>
<td>Model scaling, Coarse-to-fine head</td>
<td>TD-RD</td>
<td>85.6</td>
</tr>
<tr>
<td>YOLOv8n [<xref ref-type="bibr" rid="ref-23">23</xref>]</td>
<td>C2f, CSPDarkNet</td>
<td>Decoupled head, Strong augmentations</td>
<td>TD-RD</td>
<td>86.0</td>
</tr>
<tr>
<td>YOLOS-ti [<xref ref-type="bibr" rid="ref-24">24</xref>]</td>
<td>ViT-Tiny</td>
<td>Visual tokenization, Lightweight attention</td>
<td>TD-RD</td>
<td>85.3</td>
</tr>
<tr>
<td>RT-DETR-R18 [<xref ref-type="bibr" rid="ref-25">25</xref>]</td>
<td>ResNet-18 &#x002B; DETR Head</td>
<td>Query selection, Two-stage decoder</td>
<td>TD-RD</td>
<td>86.1</td>
</tr>
<tr>
<td>Lite-DETR [<xref ref-type="bibr" rid="ref-26">26</xref>]</td>
<td>MobileViT</td>
<td>Lightweight cross attention, Fast convergence</td>
<td>TD-RD</td>
<td>86.5</td>
</tr>
<tr>
<td>Faster R-CNN [<xref ref-type="bibr" rid="ref-27">27</xref>]</td>
<td>ResNet-50</td>
<td>Region proposal network (RPN)</td>
<td>TD-RD</td>
<td>82.7</td>
</tr>
<tr>
<td>SSD30 [<xref ref-type="bibr" rid="ref-28">28</xref>]</td>
<td>VGG-16</td>
<td>Multi-scale feature maps</td>
<td>TD-RD</td>
<td>80.3</td>
</tr>
<tr>
<td><bold>MGD-YOLO (Ours)</bold></td>
<td>YOLOv5 Backbone</td>
<td>MSDA, DSC, VGAU</td>
<td>TD-RD</td>
<td><bold>88.3</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In the context of road defect detection, attention modules help reduce background interference and emphasize texture-disruptive regions. For instance, Liu et al. [<xref ref-type="bibr" rid="ref-29">29</xref>] proposed a graph-based attention fusion method to integrate multiple defect cues. Wang et al. [<xref ref-type="bibr" rid="ref-30">30</xref>] used dual attention paths to handle noise and occlusion. Inspired by these advances, our work incorporates a Multi-Scale Dilated Attention (MSDA) module, which enables the model to simultaneously focus on semantic patterns at various receptive field sizes, effectively enhancing sensitivity to subtle structural anomalies.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Lightweight Design and Multi-Scale Feature Fusion</title>
<p>Deploying detection models in real-world infrastructure applications often requires low-latency and lightweight architectures. Depthwise Separable Convolutions, first introduced in MobileNet [<xref ref-type="bibr" rid="ref-31">31</xref>,<xref ref-type="bibr" rid="ref-32">32</xref>], have become a standard tool for reducing parameter count and computation, followed by enhancements like inverted residual blocks in MobileNetV2 [<xref ref-type="bibr" rid="ref-33">33</xref>] and re-parameterization in RepVGG [<xref ref-type="bibr" rid="ref-34">34</xref>]. For real-time edge deployment, recent frameworks like YOLOv7-Tiny [<xref ref-type="bibr" rid="ref-22">22</xref>,<xref ref-type="bibr" rid="ref-35">35</xref>,<xref ref-type="bibr" rid="ref-36">36</xref>] and YOLO-NAS [<xref ref-type="bibr" rid="ref-37">37</xref>] attempt to balance accuracy and efficiency through backbone redesign and NAS-based optimization.</p>
<p>On the feature fusion side, models such as FPN [<xref ref-type="bibr" rid="ref-38">38</xref>], PANet [<xref ref-type="bibr" rid="ref-39">39</xref>], and BiFPN [<xref ref-type="bibr" rid="ref-40">40</xref>] improve multi-scale prediction by enhancing the flow of semantic information across network layers. However, simple concatenation or summation can introduce redundancy and reduce spatial precision. To address this, attention-guided fusion modules [<xref ref-type="bibr" rid="ref-41">41</xref>&#x2013;<xref ref-type="bibr" rid="ref-44">44</xref>] and context-aware decoders [<xref ref-type="bibr" rid="ref-45">45</xref>&#x2013;<xref ref-type="bibr" rid="ref-47">47</xref>] have been proposed. Our method builds upon this line of work by designing a Visual Global Attention Upsampling (VGAU) module that leverages high-level semantics as global context guidance to refine low-level feature maps during upsampling, thereby enhancing the localization of small or low-contrast defects.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Methodology</title>
<p>Although YOLOv5 has demonstrated impressive performance in real-time object detection, it tends to rely heavily on low-level feature maps for prediction, which often leads to the loss of critical high-level semantic information. This limitation becomes particularly pronounced in multi-scale detection scenarios, where the lack of global contextual understanding undermines the accurate localization of small or subtle defects. Furthermore, the commonly used CBL block (Convolution, Batch Normalization, Leaky ReLU) in YOLOv5 employs standard convolutions, resulting in a large number of parameters and high computational overhead. These characteristics significantly restrict the model&#x2019;s deployment on resource-constrained devices, such as mobile or embedded systems.</p>
<p>To overcome these limitations, we propose MGD-YOLO, a structurally enhanced variant of YOLOv5 tailored for robust and efficient road defect detection. Our design introduces three key architectural innovations to improve the model&#x2019;s expressiveness, accuracy, and computational efficiency.</p>
<p>As illustrated in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>, MGD-YOLO retains the core structure of YOLOv5 but incorporates the following enhancements: First, we embed a Multi-Scale Dilated Attention (MSDA) module into the backbone, enabling the network to model semantic dependencies across varying receptive fields. This enhances the feature representation capacity for defects of different sizes and textures, especially in complex scenes. Second, to reduce the computational burden, we replace standard convolutions in CBL blocks with Depthwise Separable Convolution (DSC) [<xref ref-type="bibr" rid="ref-31">31</xref>], which decomposes the convolution operation into spatial and channel-wise components. This substitution significantly lowers the number of parameters and floating-point operations (FLOPs), while maintaining the model&#x2019;s expressive power. Third, we introduce a novel Visual Global Attention Upsampling (VGAU) module, which refines the low-level feature maps using global semantic cues derived from high-level features. This facilitates more precise spatial localization of defects, particularly small-scale or low-contrast anomalies that may otherwise be overlooked. These enhancements collectively improve the accuracy, robustness, and deployment efficiency of MGD-YOLO. The model is particularly well-suited for real-time road inspection applications where detection precision and computational cost must be carefully balanced. In the following subsections, we provide detailed descriptions of each module integrated into the MGD-YOLO framework.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Overall architecture of the proposed MGD-YOLO model. The backbone is enhanced with Multi-Scale Dilated Attention (MSDA), Depthwise Separable Convolutions (DSC), and Visual Global Attention Upsampling (VGAU) to improve feature fusion and detection performance</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-2.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Attention-Based Multi-Scale Feature Extraction via MSDA</title>
<p>In object detection tasks, high-level feature maps typically encapsulate rich semantic information but suffer from limited spatial resolution, making it challenging to accurately localize fine-grained targets. Conversely, low-level feature maps preserve high-resolution spatial details yet lack semantic abstraction. Bridging this semantic-resolution gap is critical for improving detection performance across varied object scales, particularly in complex scenarios such as road defect detection. Previous studies have attempted to address this challenge through hierarchical feature fusion [<xref ref-type="bibr" rid="ref-38">38</xref>,<xref ref-type="bibr" rid="ref-39">39</xref>], yet simple aggregation often introduces redundant information and fails to resolve semantic inconsistencies between layers.</p>
<p>To overcome these limitations, we incorporate an attention-based strategy into the feature fusion process of YOLOv5 by introducing the Multi-Scale Dilated Attention (MSDA) module. Attention mechanisms have shown significant effectiveness across various domains, including object detection [<xref ref-type="bibr" rid="ref-18">18</xref>], semantic segmentation [<xref ref-type="bibr" rid="ref-48">48</xref>], and natural language processing [<xref ref-type="bibr" rid="ref-49">49</xref>,<xref ref-type="bibr" rid="ref-50">50</xref>], due to their ability to dynamically emphasize task-relevant features while suppressing irrelevant noise. Classic modules such as Squeeze-and-Excitation (SE) [<xref ref-type="bibr" rid="ref-15">15</xref>], CBAM [<xref ref-type="bibr" rid="ref-18">18</xref>], and ECA [<xref ref-type="bibr" rid="ref-16">16</xref>] have demonstrated the benefit of channel-wise and spatial recalibration. However, these approaches often lack flexibility in modeling variable context scales. In contrast, MSDA introduces dilated self-attention across multiple receptive fields, enabling the network to capture both local details and global semantic dependencies in a unified framework.</p>
<p>As shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, given an input feature map <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mi>F</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, the MSDA module first splits it into multiple attention heads along the channel dimension. Each head applies a window-based self-attention mechanism with a specific dilation rate <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mi>r</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula> to expand its receptive field. This enables each head to capture context at different spatial scales. For a query position <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mi>x</mml:mi></mml:math></inline-formula>, the output of head <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mi>i</mml:mi></mml:math></inline-formula> is computed as:
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:msub><mml:mrow><mml:mtext mathvariant="bold">A</mml:mtext></mml:mrow><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:munder><mml:mrow><mml:mtext>Softmax</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>&#x03D5;</mml:mi><mml:mi>q</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mi mathvariant="normal">&#x22A4;</mml:mi></mml:msup><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>&#x03D5;</mml:mi><mml:mi>k</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>&#x03D5;</mml:mi><mml:mi>v</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>y</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Overview of the proposed Multi-Scale Dilated Attention (MSDA) module. Each head attends to features at a distinct dilation rate, aggregating multi-scale contextual information</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-3.tif"/>
</fig>
<p>where <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:msub><mml:mrow><mml:mi>&#x1D4A9;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> denotes the neighborhood region centered at <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mi>x</mml:mi></mml:math></inline-formula> under dilation <inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:msub><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula>, and <inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:msub><mml:mi>&#x03D5;</mml:mi><mml:mi>q</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:msub><mml:mi>&#x03D5;</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:math></inline-formula>, <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:msub><mml:mi>&#x03D5;</mml:mi><mml:mi>v</mml:mi></mml:msub></mml:math></inline-formula> are learnable linear projections for queries, keys, and values, respectively. The outputs of all attention heads are then concatenated and passed through a lightweight multi-layer perceptron (MLP) to produce the refined feature map:
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:msup><mml:mi>F</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mrow><mml:mtext>MLP</mml:mtext></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>Concat</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">A</mml:mtext></mml:mrow><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">A</mml:mtext></mml:mrow><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mtext mathvariant="bold">A</mml:mtext></mml:mrow><mml:mn>3</mml:mn></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>F</mml:mi></mml:math></disp-formula></p>
<p>This design enables MSDA to aggregate multi-scale semantic cues while maintaining efficiency through dilated sampling, thus reducing information redundancy without introducing additional heavy computation.</p>
<p>The combination of dilated sampling and dynamic attention aggregation not only enhances the representational capacity at multiple scales but also reduces information redundancy compared to naive multi-branch fusion, leading to improved feature discriminability with lower computational cost.</p>
<p>We integrate MSDA immediately after the C3 module within the YOLOv5 backbone, as depicted in <xref ref-type="fig" rid="fig-4">Fig. 4</xref>. This placement ensures that enriched multi-scale semantic features are incorporated before the upsampling stage. By doing so, low-level features retain detailed structural information, while high-level features provide contextual guidance, forming a more consistent and discriminative representation for defect detection.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>MSDA implementation details. The top shows the module structure consisting of DWConv, sliding windows, and MLP layers; the bottom illustrates the integration of MSDA into the feature fusion pathway</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-4.tif"/>
</fig>
<p>Theoretically, the multi-scale dilated attention promotes a more stable feature learning process by ensuring that both localized anomalies (e.g., small cracks) and broader contextual cues (e.g., surface material transitions) are simultaneously emphasized during forward and backward propagation. This stabilization effect improves model convergence behavior and robustness during training.</p>
<p>Empirically, this configuration allows the model to better focus on subtle texture changes and irregular defect boundaries that are critical in road inspection tasks. Consequently, the MSDA-enhanced feature maps significantly contribute to improving the model&#x2019;s detection accuracy, particularly in identifying small, scattered, or visually ambiguous road defects.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Visual Global Attention Upsampling (VGAU)</title>
<p>While convolutional neural networks (CNNs) have achieved remarkable success in object detection due to their hierarchical feature representation and end-to-end trainability [<xref ref-type="bibr" rid="ref-51">51</xref>,<xref ref-type="bibr" rid="ref-52">52</xref>], they often suffer from loss of fine-grained spatial details during deep feature extraction. High-level features, although semantically rich, are typically downsampled and lose precise localization cues, which is particularly detrimental for detecting small or low-contrast objects like road cracks or surface repairs. Conversely, low-level features retain spatial resolution but lack semantic context, making it challenging to distinguish defects from background textures.</p>
<p>To address this issue, various U-shaped architectures [<xref ref-type="bibr" rid="ref-53">53</xref>,<xref ref-type="bibr" rid="ref-54">54</xref>] have explored the integration of decoder paths to restore fine details. However, these designs often involve complex multi-stage decoders and impose high computational overhead, limiting their suitability for real-time applications on resource-constrained platforms.</p>
<p>To enable efficient and context-aware upsampling, we draw inspiration from the Global Attention Upsampling (GAU) module [<xref ref-type="bibr" rid="ref-55">55</xref>,<xref ref-type="bibr" rid="ref-56">56</xref>] and propose an improved variant tailored for road defect detection, termed Visual Global Attention Upsampling (VGAU). Our VGAU module leverages high-level global semantic context to recalibrate low-level feature responses, enhancing localization precision without introducing significant computational complexity. Unlike traditional decoder structures that independently process low-level features, VGAU introduces top-down semantic guidance during upsampling, which improves the discrimination ability of low-level feature activations while stabilizing feature propagation across the network. This facilitates a smoother gradient flow and more robust convergence during training.</p>
<p>As illustrated in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>, given a high-level feature map <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:msub><mml:mi>F</mml:mi><mml:mi>h</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> and a corresponding low-level feature map <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:msub><mml:mi>F</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>2</mml:mn><mml:mi>W</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mi>C</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup></mml:mrow></mml:msup></mml:math></inline-formula>, we first compute a global semantic vector via global average pooling:
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:mtext mathvariant="bold">g</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>H</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:mfrac><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>H</mml:mi></mml:mrow></mml:munderover><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>W</mml:mi></mml:mrow></mml:munderover><mml:msub><mml:mi>F</mml:mi><mml:mi>h</mml:mi></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Architecture of the proposed visual global attention upsampling (VGAU) module. High-level global context modulates low-level features through channel-wise attention, followed by upsampling and fusion</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-5.tif"/>
</fig>
<p>This global descriptor <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mrow><mml:mtext mathvariant="bold">g</mml:mtext></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is then transformed through a lightweight channel-wise excitation function composed of a <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> convolution, batch normalization, and a non-linear SiLU activation:
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>BN</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mtext mathvariant="bold">g</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<p>Meanwhile, the low-level feature map <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:msub><mml:mi>F</mml:mi><mml:mi>l</mml:mi></mml:msub></mml:math></inline-formula> is compressed using a <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula> convolution to reduce channel dimensionality. The attention-guided modulation is then applied by reweighting <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:msub><mml:mi>F</mml:mi><mml:mi>l</mml:mi></mml:msub></mml:math></inline-formula> with <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow></mml:math></inline-formula>:
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:msub><mml:mrow><mml:mover><mml:mi>F</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>l</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext mathvariant="bold">w</mml:mtext></mml:mrow><mml:mo>&#x2299;</mml:mo><mml:msub><mml:mrow><mml:mtext>Conv</mml:mtext></mml:mrow><mml:mrow><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<p>The recalibrated low-level features <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msub><mml:mrow><mml:mover><mml:mi>F</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>l</mml:mi></mml:msub></mml:math></inline-formula> are then fused with the upsampled high-level features via summation:
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mtext>Upsample</mml:mtext></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>h</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mrow><mml:mover><mml:mi>F</mml:mi><mml:mo stretchy="false">&#x007E;</mml:mo></mml:mover></mml:mrow><mml:mi>l</mml:mi></mml:msub></mml:math></disp-formula></p>
<p>Theoretically, VGAU enhances training stability by aligning low-level feature distributions with high-level semantic priors, which reduces feature noise and suppresses gradient vanishing phenomena in the decoder pathway. The use of channel-wise attention ensures that the network dynamically emphasizes important semantic clues while filtering irrelevant background patterns, thereby accelerating convergence and improving generalization performance.</p>
<p>As shown in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>, VGAU is embedded in the upsampling pathway of the MGD-YOLO architecture, where it operates in conjunction with the MSDA and DSC modules. This integration enables the network to preserve both the fine-grained spatial cues and semantic richness required for robust road defect detection.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Utilization of VGAU within MGD-YOLO: MSDA-enhanced features are progressively refined through C3, DSC, and VGAU blocks for accurate multi-scale prediction</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-6.tif"/>
</fig>
<p>Compared to the original GAU, our VGAU incorporates two major improvements: (1) the use of the SiLU activation for smoother gradient propagation and non-linearity, and (2) an additional upsampling operation post-fusion to enhance resolution alignment. Overall, VGAU not only improves multi-scale feature fusion efficiency but also enhances model training dynamics by promoting consistent semantic flow across layers, ensuring higher stability and robustness in real-world deployment. These enhancements allow the module to operate efficiently across multi-scale representations, ultimately contributing to improved precision and recall in our detection results.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Depthwise Separable Convolution</title>
<p>Traditional convolutional operations, as adopted in the original YOLOv5 architecture, compute correlations between input features and convolution kernels across both spatial and channel dimensions simultaneously. While effective in capturing local patterns, this approach introduces significant computational overhead, especially when dealing with high-dimensional feature maps. The number of parameters and the computational complexity of a standard convolutional layer are given by:
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>Parameters</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>standard</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>FLOPs</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>standard</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mi>W</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>where <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> denote the number of input and output channels, <italic>K</italic> is the kernel size, and <italic>H</italic><sup><italic>&#x2032;</italic></sup>, <italic>W</italic><sup><italic>&#x2032;</italic></sup> are the spatial dimensions of the output feature map.</p>
<p>In the context of real-time road defect detection, such computational demands pose serious limitations, especially for deployment on edge devices with constrained resources. To alleviate this issue and accelerate inference, we replace all standard convolution layers in the network with Depthwise Separable Convolution (DSC) modules, a lightweight alternative initially proposed in MobileNet [<xref ref-type="bibr" rid="ref-31">31</xref>]. As shown in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>, DSC factorizes the convolution operation into two independent steps: <italic>depthwise convolution</italic> and <italic>pointwise convolution</italic>.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Illustration of the Depthwise Separable Convolution (DSC) module, which decomposes standard convolution into channel-wise and linear projection components to reduce computation</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-7.tif"/>
</fig>
<p><bold>1) Depthwise Convolution.</bold> This step applies a spatial convolution independently to each input channel. For a kernel size <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:mi>K</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>K</mml:mi></mml:math></inline-formula>, the number of parameters and computational cost are reduced to:
<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>Parameters</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>depthwise</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>K</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>FLOPs</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>depthwise</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mi>W</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:mi>K</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>K</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p><bold>2) Pointwise Convolution.</bold> A <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> convolution is applied across channels to linearly combine the outputs from the depthwise step. Its parameter count and computation are:
<disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>Parameters</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>pointwise</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-12"><label>(12)</label><mml:math id="mml-eqn-12" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>FLOPs</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>pointwise</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x00D7;</mml:mo><mml:msup><mml:mi>W</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p><bold>3) Total Complexity.</bold> By combining both operations, the total cost of a DSC layer becomes:
<disp-formula id="eqn-13"><label>(13)</label><mml:math id="mml-eqn-13" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>Parameters</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>DSC</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msup><mml:mi>K</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="eqn-14"><label>(14)</label><mml:math id="mml-eqn-14" display="block"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>FLOPs</mml:mtext></mml:mrow><mml:mrow><mml:mrow><mml:mtext>DSC</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi></mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mi>H</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x22C5;</mml:mo><mml:msup><mml:mi>W</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msup><mml:mi>K</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mo>+</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula></p>
<p>Compared to standard convolution, this results in an approximate reduction factor of:
<disp-formula id="eqn-15"><label>(15)</label><mml:math id="mml-eqn-15" display="block"><mml:mfrac><mml:mn>1</mml:mn><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:msup><mml:mi>K</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mfrac><mml:mspace width="1em" /><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext>assuming&#xA0;</mml:mtext></mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>in</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula></p>
<p>This structural decomposition enables the network to maintain its feature learning capacity while drastically reducing both parameter count and floating-point operations. Such efficiency gains are particularly valuable in road defect detection, where real-time performance and lightweight deployment are crucial.</p>
<p>Moreover, the use of DSC enhances the model&#x2019;s generalization ability by limiting overfitting from redundant parameterization and encourages efficient representation learning. In our MGD-YOLO architecture, DSC replaces all conventional CBL (Convolution &#x002B; BatchNorm &#x002B; LeakyReLU) blocks, further improving inference speed and making the model well-suited for deployment on embedded or mobile platforms for large-scale road condition monitoring.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Why YOLOv5 as the Baseline?</title>
<p>Although more recent models in the YOLO series&#x2014;such as YOLOv8 [<xref ref-type="bibr" rid="ref-23">23</xref>], YOLOv9 [<xref ref-type="bibr" rid="ref-57">57</xref>], and YOLOv10 [<xref ref-type="bibr" rid="ref-58">58</xref>]&#x2014;offer improvements in detection accuracy and architectural novelty, we choose YOLOv5 as the baseline framework for MGD-YOLO primarily due to its maturity, stability, and deployability in real-world scenarios. As our target application emphasizes real-time defect detection on vehicle-mounted edge devices, lightweight design and efficient inference are of paramount importance. YOLOv5 strikes a practical balance between accuracy and computational cost, with a modular architecture that facilitates easy customization and integration of new components such as MSDA, DSC, and VGAU. In contrast, newer versions often increase model complexity and hardware requirements, which may hinder their deployment in resource-constrained environments. Furthermore, YOLOv5 remains a widely accepted baseline in many road defect detection benchmarks, enabling consistent and fair comparisons with prior work. By enhancing YOLOv5 with carefully designed modules, we demonstrate that significant performance gains can be achieved without sacrificing speed or portability, making the model more suitable for intelligent transportation systems and edge computing platforms.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiment</title>
<sec id="s4_1">
<label>4.1</label>
<title>Dataset Preparation and Experimental Environment</title>
<p>To comprehensively evaluate the effectiveness of the proposed MGD-YOLO framework, we conducted experiments on three publicly available road defect detection datasets: TD-RD, CNRDD, and CRDDC&#x2019;22. These datasets collectively include various types of road surfaces&#x2013;such as cement and asphalt&#x2013;and cover three representative categories of surface anomalies: <italic>cracks</italic>, <italic>repairs</italic>, and <italic>potholes</italic>. All images were uniformly resized to a resolution of <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mn>640</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>640</mml:mn></mml:math></inline-formula> pixels to ensure consistency during model training and inference. <xref ref-type="fig" rid="fig-8">Fig. 8</xref> provides representative samples from these datasets.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Representative examples of road surface defects: (<bold>a</bold>) crack, (<bold>b</bold>) repair, (<bold>c</bold>) pothole</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-8.tif"/>
</fig>
<p>Specifically, TD-RD contains 1532 annotated images collected from three cities in China, CNRDD includes 4218 images from multiple provinces, and CRDDC&#x2019;22 consists of 9301 images gathered across five countries. We did not adopt the RDD2022 dataset because of its significant label imbalance and annotation inconsistencies, which could introduce noise into model training.</p>
<p>Each dataset was split into training, validation, and test sets using a 60:20:20 ratio. All annotations were provided in YOLO format or converted accordingly, and we used the LabelImg tool for any necessary modifications or corrections. The specific road defect types included in each dataset are summarized in <xref ref-type="table" rid="table-2">Table 2</xref>.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Summary of road defect types in each dataset</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th align="center">Dataset</th>
<th align="center">Defect types</th>
</tr>
</thead>
<tbody>
<tr>
<td>TD-RD</td>
<td>Crack, Repair, Pothole</td>
</tr>
<tr>
<td>CNRDD</td>
<td>Crack, Pothole, Surface wear</td>
</tr>
<tr>
<td>CRDDC&#x2019;22</td>
<td>Fine crack, Wide crack, Patch edge, Pothole, Surface abrasion</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>All experiments were conducted on a Windows 10 workstation equipped with an Intel Core i9-10900K CPU and an NVIDIA A100 80GB. The MGD-YOLO model was implemented in PyTorch and trained for 200 epochs with a batch size of 16. All settings aligned with the TD-RD. To enhance generalization, we employed standard data augmentation techniques including mosaic augmentation, random scaling, and horizontal flipping. We also fixed a random seed to ensure reproducibility of results and repeated each experiment three times to report averaged performance.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Comparison with State-of-the-Art Methods across Benchmarks</title>
<p>To thoroughly assess the effectiveness and efficiency of our proposed MGD-YOLO (TD-YOLOv10) framework, we conducted comparative experiments against a wide range of state-of-the-art object detectors, including both CNN-based models (e.g., YOLOv5/6/7/8/9/10 series, PP-PicoDet) and Transformer-based architectures (e.g., YOLOS, RT-DERT, Lite-DERT). The evaluation was performed on three representative road defect detection datasets: TD-RD [<xref ref-type="bibr" rid="ref-59">59</xref>], CNRDD, and CRDDC&#x2019;22.</p>
<p><xref ref-type="table" rid="table-3">Table 3</xref> summarizes the results in terms of mean average precision (mAP), precision (Pre), computational cost (FLOPs), and inference speed (FPS). The best results are highlighted in <bold>bold</bold>, the second best in red, and the third best in blue.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Performance comparison with state-of-the-art models across three road defect datasets. Best results are <bold>bold</bold>, second-best in red, third-best in blue</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
</colgroup>
<thead>
<tr>
<th rowspan="2" align="center">Model</th>
<th colspan="4">TD-RD</th>
<th colspan="4">CNRDD</th>
<th colspan="4">CRDDC<bold>&#x2019;</bold>22</th>
</tr>
<tr>
<th align="center">mAP <bold>(%)</bold></th>
<th align="center">Pre <bold>(%)</bold></th>
<th align="center">FLOPs</th>
<th align="center">FPS</th>
<th align="center">mAP <bold>(%)</bold></th>
<th align="center">Pre <bold>(%)</bold></th>
<th align="center">FLOPs</th>
<th align="center">FPS</th>
<th align="center">mAP <bold>(%)</bold></th>
<th align="center">Pre <bold>(%)</bold></th>
<th align="center">FLOPs</th>
<th align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td>YOLOv5-n</td>
<td>81.4</td>
<td>79.8</td>
<td><bold>4.10</bold></td>
<td>139</td>
<td>21.4</td>
<td>33.7</td>
<td><bold>4.10</bold></td>
<td>139</td>
<td>41.4</td>
<td>44.7</td>
<td><bold>4.10</bold></td>
<td>139</td>
</tr>
<tr>
<td>YOLOv5-s</td>
<td>85.6</td>
<td>84.6</td>
<td>15.8</td>
<td>111</td>
<td>22.5</td>
<td>30.7</td>
<td>15.8</td>
<td>111</td>
<td>42.1</td>
<td>46.4</td>
<td>15.8</td>
<td>111</td>
</tr>
<tr>
<td>YOLOv6-n [<xref ref-type="bibr" rid="ref-21">21</xref>][arXiv&#x2019;22]</td>
<td>78.3</td>
<td>76.9</td>
<td>11.4</td>
<td>123</td>
<td>21.4</td>
<td>31.8</td>
<td>11.4</td>
<td>123</td>
<td>42.1</td>
<td>46.4</td>
<td>11.4</td>
<td>123</td>
</tr>
<tr>
<td>YOLOv6-s [<xref ref-type="bibr" rid="ref-21">21</xref>][arXiv&#x2019;22]</td>
<td>83.0</td>
<td>82.5</td>
<td>45.3</td>
<td>81</td>
<td>24.6</td>
<td>33.8</td>
<td>45.3</td>
<td>81</td>
<td>42.4</td>
<td>46.0</td>
<td>45.3</td>
<td>81</td>
</tr>
<tr>
<td>YOLOv7-ti [<xref ref-type="bibr" rid="ref-22">22</xref>][CVPR&#x2019;23]</td>
<td>84.5</td>
<td>85.7</td>
<td>13.2</td>
<td>294</td>
<td>25.3</td>
<td>33.5</td>
<td>13.2</td>
<td>294</td>
<td>46.2</td>
<td>49.8</td>
<td>13.2</td>
<td>294</td>
</tr>
<tr>
<td>YOLOv8-n [<xref ref-type="bibr" rid="ref-23">23</xref>][arXiv&#x2019;24]</td>
<td>82.2</td>
<td>81.9</td>
<td>8.2</td>
<td><bold>385</bold></td>
<td>27.6</td>
<td>38.4</td>
<td>8.2</td>
<td><bold>385</bold></td>
<td>46.0</td>
<td>48.5</td>
<td>8.2</td>
<td><bold>385</bold></td>
</tr>
<tr>
<td>YOLOv8-s [<xref ref-type="bibr" rid="ref-23">23</xref>][arXiv&#x2019;24]</td>
<td>85.1</td>
<td>86.0</td>
<td>28.4</td>
<td>333</td>
<td>27.6</td>
<td>38.4</td>
<td>28.4</td>
<td>333</td>
<td>46.0</td>
<td>48.5</td>
<td>28.4</td>
<td>333</td>
</tr>
<tr>
<td>YOLOv9-s [<xref ref-type="bibr" rid="ref-57">57</xref>][arXiv&#x2019;24]</td>
<td>85.2</td>
<td><bold>88.6</bold></td>
<td>30.3</td>
<td>172</td>
<td>29.5</td>
<td>37.4</td>
<td>30.3</td>
<td>172</td>
<td>47.4</td>
<td>49.7</td>
<td>30.3</td>
<td>172</td>
</tr>
<tr>
<td>YOLOv10-n [<xref ref-type="bibr" rid="ref-58">58</xref>][arXiv&#x2019;24]</td>
<td>82.3</td>
<td>81.4</td>
<td>8.22</td>
<td>357</td>
<td>28.1</td>
<td>35.9</td>
<td>8.22</td>
<td>357</td>
<td>46.5</td>
<td>48.3</td>
<td>8.22</td>
<td>357</td>
</tr>
<tr>
<td>YOLOv10-s [<xref ref-type="bibr" rid="ref-58">58</xref>] [arXiv&#x2019;24]</td>
<td>85.0</td>
<td>82.2</td>
<td>24.5</td>
<td>286</td>
<td>28.8</td>
<td>39.4</td>
<td>24.5</td>
<td>286</td>
<td>47.3</td>
<td><bold>57.3</bold></td>
<td>24.5</td>
<td>286</td>
</tr>
<tr>
<td>YOLOS-ti [<xref ref-type="bibr" rid="ref-24">24</xref>][arXiv&#x2019;21]</td>
<td>80.8</td>
<td>80.4</td>
<td>21</td>
<td>116</td>
<td>21.3</td>
<td>30.1</td>
<td>21</td>
<td>116</td>
<td>45.3</td>
<td>52.0</td>
<td>24.5</td>
<td>286</td>
</tr>
<tr>
<td>YOLOS-s [<xref ref-type="bibr" rid="ref-24">24</xref>][arXiv&#x2019;21]</td>
<td>84.7</td>
<td>83.2</td>
<td>179</td>
<td>54</td>
<td>23.6</td>
<td>36.4</td>
<td>179</td>
<td>54</td>
<td>46.8</td>
<td>49.4</td>
<td>179</td>
<td>54</td>
</tr>
<tr>
<td>PP-PicoDet [<xref ref-type="bibr" rid="ref-60">60</xref>][arXiv&#x2019;21]</td>
<td>85.6</td>
<td>83.4</td>
<td>8.9</td>
<td>196</td>
<td>22.4</td>
<td>31.7</td>
<td>8.9</td>
<td>196</td>
<td>47.0</td>
<td>48.0</td>
<td>8.9</td>
<td>196</td>
</tr>
<tr>
<td>RT-DERT [<xref ref-type="bibr" rid="ref-25">25</xref>][CVPR&#x2019;23]</td>
<td>87.7</td>
<td>87.7</td>
<td>60</td>
<td>159</td>
<td>29.4</td>
<td>39.5</td>
<td>60</td>
<td>159</td>
<td><bold>48.6</bold></td>
<td>51.7</td>
<td>60</td>
<td>159</td>
</tr>
<tr>
<td>Lite-DERT [<xref ref-type="bibr" rid="ref-26">26</xref>] [CVPR&#x2019;21]</td>
<td>86.1</td>
<td>85.2</td>
<td>151</td>
<td>75</td>
<td>26.3</td>
<td>33.0</td>
<td>151</td>
<td>75</td>
<td>45.3</td>
<td>48.9</td>
<td>151</td>
<td>75</td>
</tr>
<tr>
<td>FR-CNN [<xref ref-type="bibr" rid="ref-27">27</xref>] [NIPS&#x2019;15]</td>
<td>74.6</td>
<td>76.6</td>
<td>94.3</td>
<td>10</td>
<td>20.3</td>
<td>36.3</td>
<td>94.3</td>
<td>10</td>
<td>39.9</td>
<td>46.1</td>
<td>94.3</td>
<td>10</td>
</tr>
<tr>
<td>SSD-VGG16 [<xref ref-type="bibr" rid="ref-28">28</xref>] [ECCV&#x2019;16]</td>
<td>66.5</td>
<td>71.1</td>
<td>60.9</td>
<td>14</td>
<td>18.5</td>
<td><bold>49.9</bold></td>
<td>60.9</td>
<td>14</td>
<td>38.7</td>
<td>46.2</td>
<td>60.9</td>
<td>14</td>
</tr>
<tr>
<td><bold>MGD-YOLOv10 (Ours)</bold></td>
<td><bold>87.9</bold></td>
<td>88.3</td>
<td>33.6</td>
<td>240</td>
<td><bold>36.2</bold></td>
<td>46.0</td>
<td>33.6</td>
<td>240</td>
<td>47.6</td>
<td>53.8</td>
<td>33.6</td>
<td>240</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As seen in the table, our MGD-YOLO consistently achieves top-tier performance across all benchmarks. Specifically, on the TD-RD dataset, it delivers the highest mAP of <bold>87.9%</bold>, significantly outperforming models such as YOLOv9-s (85.2%) and RT-DERT (87.7%). In terms of inference speed, MGD-YOLO runs at 240 FPS, which is competitive with lightweight models like YOLOv8-n and YOLOv10-n, while maintaining superior detection accuracy.</p>
<p>Across the CNRDD and CRDDC&#x2019;22 datasets, MGD-YOLO also demonstrates strong generalization ability, achieving the highest or second-best scores in both mAP and precision. Notably, on CRDDC&#x2019;22, it reaches a precision of 53.8%, narrowly trailing the best model in that metric while outperforming all others in speed and FLOPs efficiency.</p>
<p>These results highlight the capability of MGD-YOLO to strike a fine balance between detection accuracy, inference efficiency, and deployment readiness&#x2013;making it well-suited for real-time road defect detection in both cloud-based and edge-based environments.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Qualitative Results and Visual Analysis</title>
<p>To further assess the interpretability and effectiveness of MGD-YOLO, we conducted qualitative visualizations including feature space distribution via t-SNE and attention heatmaps from the VGAU module.</p>
<p><bold>Comparison with Transformer-Based Detectors.</bold> In addition to YOLO-based baselines, we compared MGD-YOLO against lightweight transformer-based detectors such as RT-DETR-R18 and Lite-DETR. To ensure fairness, we selected configurations with similar FLOPs and parameter scales to MGD-YOLO. As shown in <xref ref-type="table" rid="table-4">Table 4</xref>, MGD-YOLO consistently outperforms these models on the TD-RD dataset, demonstrating both higher accuracy and better inference speed.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Comparison with transformer-based detectors (Similar FLOPs/Params)</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Model</th>
<th>FLOPs (G)</th>
<th>Params (M)</th>
<th><inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:msub><mml:mtext>mAP</mml:mtext><mml:mrow><mml:mn mathvariant="bold">50</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> <bold>(%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td>RT-DETR-R18</td>
<td>24.8</td>
<td>36.7</td>
<td>86.1</td>
</tr>
<tr>
<td>Lite-DETR</td>
<td>22.3</td>
<td>32.5</td>
<td>86.5</td>
</tr>
<tr>
<td><bold>MGD-YOLO (Ours)</bold></td>
<td>23.5</td>
<td>34.2</td>
<td><bold>88.3</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p><bold>Feature Embedding Visualization.</bold> We utilized t-distributed Stochastic Neighbor Embedding (t-SNE) to project high-dimensional features extracted from the penultimate layer of different models onto a 2D space. As shown in <xref ref-type="fig" rid="fig-9">Fig. 9</xref>, MGD-YOLO produces more compact and well-separated clusters for each defect class, indicating superior discriminative capability in feature learning compared to the baseline.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>t-SNE visualization of feature embeddings extracted from the final detection layer across different models based on the TD-RD dataset. Compared to YOLOv5, YOLOv10, and RT-DERT, our MGD-YOLO exhibits clearer class separation and tighter intra-class clustering, indicating stronger feature discriminability</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-9.tif"/>
</fig>
<p><bold>Attention Map Visualization.</bold> We further visualized the attention responses from the VGAU module to understand how the model focuses on defect regions. Qualitative results in <xref ref-type="fig" rid="fig-9">Fig. 9</xref> further illustrate that MGD-YOLO yields more complete and accurate detections, particularly for small and ambiguous defects. As illustrated in <xref ref-type="fig" rid="fig-10">Fig. 10</xref>, MGD-YOLO demonstrates stronger spatial localization capability by highlighting regions with fine-grained details, such as hairline cracks and boundary edges of potholes, which are often missed by other models. As shown in <xref ref-type="fig" rid="fig-11">Fig. 11</xref>, our model outperforms other baselines.</p>
<fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>Qualitative comparison of attention heatmaps generated by YOLOv10, RT-DERT, and our MGD-YOLO on a road crack image. While YOLOv10 and RT-DERT produce concentrated but limited attention around the central crack region, our method captures both global and fine-grained details, accurately attending to multiple critical areas along the defect</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-10.tif"/>
</fig><fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>Qualitative comparison of detection results among YOLOv10 (<bold>a</bold>), RT-DERT (<bold>b</bold>), and our proposed MGD-YOLO (<bold>c</bold>) across multiple road defect scenarios. MGD-YOLO demonstrates superior localization and robustness, particularly in challenging cases with complex textures, shadows, or small-scale defects. It consistently identifies multiple instances with higher confidence while minimizing false positives and missed detections</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-11.tif"/>
</fig>
<p>These qualitative results corroborate our quantitative findings and demonstrate that MGD-YOLO not only improves detection accuracy but also enhances feature representation and localization precision.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Ablation Study</title>
<p>To further validate the individual contributions of each module in the proposed MGD-YOLO framework, we conducted a series of ablation experiments focusing on the three core components: Multi-Scale Dilated Attention (MSDA), Depthwise Separable Convolution (DSC), and the Visual Global Attention Upsampling (VGAU) module. <xref ref-type="table" rid="table-5">Table 5</xref> summarizes the detection performance under various module configurations on the road defect dataset.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Ablation results using combinations of MSDA (M), DSC (D), and VGAU (G) on the road defect dataset</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Model variant</th>
<th>Precision <bold>(%)</bold></th>
<th>Recall <bold>(%)</bold></th>
<th><inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:msub><mml:mtext>mAP</mml:mtext><mml:mrow><mml:mn mathvariant="bold">50</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> <bold>(%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td>YOLOv5 (Baseline)</td>
<td>81.4</td>
<td>78.7</td>
<td>81.3</td>
</tr>
<tr>
<td>M-YOLO (with MSDA)</td>
<td>80.9</td>
<td>76.1</td>
<td>82.4</td>
</tr>
<tr>
<td>D-YOLO (with DSC)</td>
<td>82.9</td>
<td>75.9</td>
<td>82.4</td>
</tr>
<tr>
<td>G-YOLO (with VGAU)</td>
<td>85.2</td>
<td>76.3</td>
<td>81.6</td>
</tr>
<tr>
<td>MD-YOLO (with MSDA &#x002B; DSC)</td>
<td>82.3</td>
<td>74.6</td>
<td>82.0</td>
</tr>
<tr>
<td><bold>MGD-YOLO (Full model)</bold></td>
<td><bold>88.3</bold></td>
<td><bold>80.3</bold></td>
<td><bold>87.9</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The experimental results highlight the effectiveness of each module. Specifically, the incorporation of DSC improves precision significantly, suggesting its utility in enhancing feature representation efficiency. MSDA proves beneficial for increasing mAP, although it results in a slight drop in recall when used in isolation. On the other hand, VGAU introduces a strong gain in precision (&#x002B;4.0%) and contributes to better localization of fine-grained defect regions.</p>
<p>When all three modules are integrated into MGD-YOLO, the model achieves the highest overall performance, with an 6.9% increase in precision, a 1.6% improvement in recall, and a 6.6% gain in <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msub><mml:mtext>mAP</mml:mtext><mml:mrow><mml:mn>50</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> over the original YOLOv5 baseline.</p>
<p>The training dynamics, illustrated in <xref ref-type="fig" rid="fig-12">Fig. 12</xref>, show that MGD-YOLO converges faster and more stably than its counterparts, while also achieving higher final accuracy.</p>
<fig id="fig-12">
<label>Figure 12</label>
<caption>
<title>Training progression of MGD-YOLO on the road defect dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-12.tif"/>
</fig>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Deployment Considerations</title>
<p>In real-world applications, deployment efficiency on different hardware platforms is a critical factor for road defect detection systems. To evaluate the practical deployability of MGD-YOLO, we conducted inference speed and memory usage tests on three representative hardware environments: NVIDIA A100 GPU (datacenter server grade), NVIDIA RTX 4090 GPU (consumer high-end grade), and NVIDIA Jetson Xavier NX (embedded edge device).</p>
<p>On the A100 GPU, MGD-YOLO achieved an average inference speed of 240 FPS with a peak memory usage of 3.2 GB. On the RTX 4090, the model achieved 215 FPS while maintaining a memory usage of 2.7 GB. On the Jetson Xavier NX, after TensorRT optimization and model pruning, MGD-YOLO maintained a real-time performance of approximately 45 FPS with a memory footprint of 1.8 GB.</p>
<p>These results demonstrate that MGD-YOLO strikes a favorable trade-off between detection accuracy and computational efficiency, enabling deployment across a wide spectrum of hardware platforms&#x2013;from high-performance servers to resource-constrained edge devices. Notably, the integration of Depthwise Separable Convolution (DSC) and Visual Global Attention Upsampling (VGAU) significantly contributes to the reduction of model size and inference latency without sacrificing accuracy.</p>
<p>Therefore, MGD-YOLO offers a flexible and scalable solution for intelligent road maintenance applications, supporting both cloud-based large-scale monitoring and decentralized on-vehicle real-time inspection systems. Future work will further optimize model quantization and pruning strategies to enhance deployment efficiency on ultra-low-power embedded systems.</p>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Misclassification Analysis</title>
<p>Although MGD-YOLO demonstrates strong overall detection performance, some misclassification cases were observed, particularly between visually similar road defect types. To better understand these errors, we conducted a qualitative analysis on the TD-RD, CNRDD, and CRDDC&#x2019;22 datasets.</p>
<p>We found that hairline cracks are occasionally confused with patch edges or surface texture artifacts, especially under poor lighting or complex backgrounds. For instance, in low-resolution images or heavily textured asphalt surfaces, small cracks may be misidentified as material joints or construction patches. Similarly, certain pothole boundaries with gradual depth transitions were sometimes mistaken for repaired areas with minor surface degradation.</p>
<p>Representative examples of such misclassifications are illustrated in <xref ref-type="fig" rid="fig-13">Fig. 13</xref>. These cases highlight the inherent difficulty in distinguishing fine-grained defect boundaries based solely on visual appearance, especially when spatial scale and intensity contrast are minimal.</p>
<fig id="fig-13">
<label>Figure 13</label>
<caption>
<title>Examples of misclassification cases observed on the road defect datasets. Small cracks and patch edges exhibit significant visual similarity under certain conditions</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_66188-fig-13.tif"/>
</fig>
<p>To address these challenges, several potential strategies are considered for future enhancement: (1) introducing multi-scale post-processing techniques to refine defect boundaries at different spatial resolutions, and (2) incorporating complementary sensing modalities such as infrared imagery or 3D surface profiling data to provide additional discriminative cues beyond RGB textures.</p>
<p>We plan to explore these directions in future work, aiming to further boost the detection accuracy and robustness of MGD-YOLO, particularly for small, low-contrast, or visually ambiguous road defects in diverse real-world environments.</p>
</sec>
<sec id="s4_7">
<label>4.7</label>
<title>Sensitivity Analysis</title>
<p>To evaluate the robustness of MGD-YOLO under different experimental settings, we conducted a brief sensitivity analysis focusing on two factors: input image resolution and dataset split ratio.</p>
<p><bold>Image Resolution:</bold> We varied the input size between <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mn>512</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>512</mml:mn></mml:math></inline-formula>, <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:mn>640</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>640</mml:mn></mml:math></inline-formula> (default), and <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mn>768</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>768</mml:mn></mml:math></inline-formula> pixels. As shown in <xref ref-type="table" rid="table-6">Table 6</xref>, MGD-YOLO maintained stable performance, with <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:msub><mml:mtext>mAP</mml:mtext><mml:mrow><mml:mn>50</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> fluctuating within 1.2% across resolutions, demonstrating resilience to input scale changes.</p>
<table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>Performance under different input resolutions</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Resolution</th>
<th><inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:msub><mml:mtext>mAP</mml:mtext><mml:mrow><mml:mn mathvariant="bold">50</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> <bold>(%)</bold></th>
<th>FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td>512 <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 512</td>
<td>85.1</td>
<td>120</td>
</tr>
<tr>
<td>640 <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 640</td>
<td>85.7</td>
<td>105</td>
</tr>
<tr>
<td>768 <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 768</td>
<td>86.2</td>
<td>92</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><bold>Dataset Split Ratio:</bold> We tested different training/validation/test splits, specifically 70/15/15 and 60/20/20. As summarized in <xref ref-type="table" rid="table-7">Table 7</xref>, the model exhibited less than 1.0% variation in <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:msub><mml:mtext>mAP</mml:mtext><mml:mrow><mml:mn>50</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, indicating good generalization under different data partitions.</p>
<table-wrap id="table-7">
<label>Table 7</label>
<caption>
<title>Performance under different data splits</title>
</caption>
<table>
<colgroup>
<col/>
<col/>
</colgroup>
<thead>
<tr>
<th>Split ratio</th>
<th><inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:msub><mml:mtext>mAP</mml:mtext><mml:mrow><mml:mn mathvariant="bold">50</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> <bold>(%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td>70/15/15</td>
<td>85.9</td>
</tr>
<tr>
<td>60/20/20</td>
<td>85.7</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>These results verify that MGD-YOLO maintains robust detection performance across varying resolutions and dataset splits, supporting its practical deployment under diverse operational conditions.</p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusion and Future Work</title>
<sec id="s5_1">
<label>5.1</label>
<title>Conclusion</title>
<p>This paper presents MGD-YOLO, an improved object detection framework based on YOLOv5, specifically designed for accurate and efficient road defect detection. By incorporating Multi-Scale Dilated Attention (MSDA), Visual Global Attention Upsampling (VGAU), and Depthwise Separable Convolution (DSC), the proposed model significantly enhances feature extraction, contextual reasoning, and computational efficiency. Extensive experiments on three public road defect datasets demonstrate that MGD-YOLO outperforms state-of-the-art models in both detection accuracy and inference speed. The model achieves superior performance in identifying diverse defect types&#x2013;including cracks, potholes, and repairs&#x2014;while maintaining a lightweight architecture suitable for real-time applications. Qualitative visualizations and ablation studies further confirm the effectiveness of each proposed component. In addition, overfitting was carefully monitored during training through validation loss tracking, early stopping, and standard data augmentation strategies.</p>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Future Work</title>
<p>In future work, we aim to further optimize MGD-YOLO by exploring lightweight backbone alternatives and neural architecture search techniques to reduce the model&#x2019;s complexity without compromising detection accuracy. Specifically, we intend to investigate the integration of efficient transformer-based modules or dynamic convolution operators to further enhance multi-scale feature extraction while maintaining low computational overhead.</p>
<p>We also plan to extend our method to multi-modal data settings, incorporating complementary cues such as thermal or LiDAR information to enhance robustness under adverse environmental conditions. Incorporating heterogeneous sensing modalities will allow MGD-YOLO to better capture subtle surface anomalies and environmental context, improving detection performance under low-visibility conditions such as nighttime, rain, or dust.</p>
<p>Additionally, we will explore domain adaptation strategies to improve generalization across different geographic regions, pavement materials, and lighting variations. We are particularly interested in adopting invariant representation learning techniques and domain adversarial training frameworks to minimize generalization error when transferring the model to new domains with distinct feature distributions.</p>
<p>Furthermore, we plan to conduct systematic sensitivity analyses on data resolution, dataset split ratios, and sensor variations to evaluate the robustness of the model under diverse operational settings, ensuring its reliability and stability during real-world deployment.</p>
<p>Ultimately, our goal is to deploy MGD-YOLO in edge devices and smart transportation systems to enable large-scale, real-time road condition monitoring in the wild. To support practical deployment, we will also benchmark the model&#x2019;s performance and resource consumption across different hardware platforms, including mobile GPUs and embedded systems, providing comprehensive guidelines for hardware-software co-optimization.</p>
</sec>
</sec>
</body>
<back>
<ack>
<p>We gratefully acknowledge the computational resources provided by the University of Alabama at Birmingham IT-Research Computing Group for High-Performance Computing (HPC) support and CPU time on the Cheaha compute cluster, which was essential for the completion of this research.</p>
</ack>
<sec>
<title>Funding Statement</title>
<p>This research was supported by Chengdu Jincheng College under the General Research Project Program (Project No. JG2024-1199), titled &#x201C;Research on the Training Mechanism of Undergraduate Innovation Ability Based on Deep Integration of AI Industry-Education Collaboration&#x201D;. The project was led by Zhengji Li and conducted as part of the Innovation Competition.</p>
</sec>
<sec>
<title>Author Contributions</title>
<p>The authors confirm contribution to the paper as follows: Conceptualization, Zhengji Li and Hao Xu; methodology, Zhengji Li; software, Zhengji Li and Boyun Huang; validation, Zhengji Li, Fazhan Xiong, and Yingrui Ji; formal analysis, Zhengji Li; investigation, Zhengji Li and Meihui Li; resources, Zhengji Li and Aokun Liang; data curation, Zhengji Li and Xi Xiao; writing&#x2014;original draft preparation, Zhengji Li; writing&#x2014;review and editing, Hao Xu and Jiacheng Xie; visualization, Zhengji Li and Fazhan Xiong; supervision, Hao Xu and Jiacheng Xie; project administration, Hao Xu; funding acquisition, Hao Xu. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials</title>
<p>The data that support the findings of this study are available from the Corresponding Author, Hao Xu, upon reasonable request.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Canny</surname> <given-names>J</given-names></string-name></person-group>. <article-title>A computational approach to edge detection</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. <year>1986</year>;<volume>8</volume>(<issue>6</issue>):<fpage>679</fpage>&#x2013;<lpage>98</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TPAMI.1986.4767851</pub-id>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>B</given-names></string-name></person-group>. <article-title>Adaptive wavelet thresholding for image denoising and compression</article-title>. <source>IEEE Trans Image Process</source>. <year>2000</year>;<volume>9</volume>(<issue>9</issue>):<fpage>1532</fpage>&#x2013;<lpage>46</lpage>. doi:<pub-id pub-id-type="doi">10.1109/83.861857</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Otsu</surname> <given-names>N</given-names></string-name></person-group>. <article-title>A threshold selection method from gray-level histograms</article-title>. <source>IEEE Trans Syst Man Cybern</source>. <year>1979</year>;<volume>9</volume>(<issue>1</issue>):<fpage>62</fpage>&#x2013;<lpage>6</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TSMC.1979.4310076</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kulambayev</surname> <given-names>B</given-names></string-name>, <string-name><surname>Beissenova</surname> <given-names>G</given-names></string-name>, <string-name><surname>Katayev</surname> <given-names>N</given-names></string-name>, <string-name><surname>Abduraimova</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhaidakbayeva</surname> <given-names>L</given-names></string-name>, <string-name><surname>Sarbassova</surname> <given-names>A</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>A deep learning-based approach for road surface damage detection</article-title>. <source>Comput Mat Contin</source>. <year>2022</year>;<volume>73</volume>(<issue>2</issue>):<fpage>3403</fpage>&#x2013;<lpage>18</lpage>. doi:<pub-id pub-id-type="doi">10.32604/cmc.2022.029544</pub-id>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Gan</surname> <given-names>X</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shim</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Road damage detection and classification using Mask R-CNN with DenseNet backbone</article-title>. <source>Comput Mat Contin</source>. <year>2020</year>;<volume>65</volume>(<issue>3</issue>):<fpage>2201</fpage>&#x2013;<lpage>15</lpage>. doi:<pub-id pub-id-type="doi">10.32604/cmc.2020.011191</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>F</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Road crack detection using deep convolutional neural network</article-title>. In: <conf-name>2016 IEEE International Conference on Image Processing (ICIP)</conf-name>; <year>2016 Sep 25&#x2013;28</year>; <publisher-loc>Phoenix, AZ, USA</publisher-loc>. p. <fpage>3708</fpage>&#x2013;<lpage>12</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICIP.2016.7532992</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Shi</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Cui</surname> <given-names>L</given-names></string-name>, <string-name><surname>Qi</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Meng</surname> <given-names>F</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>Automatic road crack detection using random structured forests</article-title>. <source>IEEE Trans Intell Transp Syst</source>. <year>2016</year>;<volume>17</volume>(<issue>12</issue>):<fpage>3434</fpage>&#x2013;<lpage>45</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TITS.2016.2569441</pub-id>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Park</surname> <given-names>S</given-names></string-name>, <string-name><surname>Bang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>H</given-names></string-name>, <string-name><surname>Kim</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Patch-based crack detection in black box images using convolutional neural networks</article-title>. <source>J Comput Civ Eng</source>. <year>2019</year>;<volume>33</volume>(<issue>3</issue>):<fpage>04019017</fpage>. doi:<pub-id pub-id-type="doi">10.1061/(ASCE)CP.1943-5487.0000831</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Jocher</surname> <given-names>G</given-names></string-name>, <string-name><surname>Chaurasia</surname> <given-names>A</given-names></string-name>, <string-name><surname>Qiu</surname> <given-names>J</given-names></string-name></person-group>. <article-title>YOLOv5 by Ultralytics; 2020 [software]. [cited 2025 Jun 4]</article-title>. Available from: <ext-link ext-link-type="uri" xlink:href="https://github.com/ultralytics/yolov5">https://github.com/ultralytics/yolov5</ext-link>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Luo</surname> <given-names>H</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Cai</surname> <given-names>L</given-names></string-name></person-group>. <article-title>An enhanced lightweight network for road damage detection based on deep learning</article-title>. <source>Electronics</source>. <year>2023</year>;<volume>12</volume>(<issue>12</issue>):<fpage>2583</fpage>. doi:<pub-id pub-id-type="doi">10.3390/electronics12122583</pub-id>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ganesh</surname> <given-names>N</given-names></string-name>, <string-name><surname>Shankar</surname> <given-names>R</given-names></string-name>, <string-name><surname>Mahdal</surname> <given-names>M</given-names></string-name>, <string-name><surname>Murugan</surname> <given-names>JS</given-names></string-name>, <string-name><surname>Chohan</surname> <given-names>JS</given-names></string-name>, <string-name><surname>Kalita</surname> <given-names>K</given-names></string-name></person-group>. <article-title>Exploring deep learning methods for computer vision applications across multiple sectors: challenges and future trends</article-title>. <source>Comput Model Eng Sci</source>. <year>2024</year>;<volume>139</volume>(<issue>1</issue>):<fpage>1</fpage>&#x2013;<lpage>28</lpage>. doi:<pub-id pub-id-type="doi">10.32604/cmes.2023.028018</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xiao</surname> <given-names>X</given-names></string-name>, <string-name><surname>Tao</surname> <given-names>L</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>K</given-names></string-name></person-group>. <article-title>An image data augmentation algorithm based on YOLOv5s-DA for pavement distress detection</article-title>. In: <conf-name>2022 5th International Conference on Pattern Recognition and Artificial Intelligence (PRAI)</conf-name>; <year>2022 Aug 19&#x2013;21</year>; <publisher-loc>Chengdu, China</publisher-loc>. p. <fpage>891</fpage>&#x2013;<lpage>5</lpage>. doi:<pub-id pub-id-type="doi">10.1109/PRAI55851.2022.9904187</pub-id>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Xiao</surname> <given-names>X</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>J</given-names></string-name>, <string-name><surname>Fan</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>G</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Cycle-YOLO: a efficient and robust framework for pavement damage detection</article-title>. <comment>arXiv:2405.17905. 2024</comment>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Chen</surname> <given-names>H</given-names></string-name>, <string-name><surname>Xue</surname> <given-names>K</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>YOLO-Pavement: an enhanced YOLOv5-based road damage detection framework with structure-aware learning</article-title>. In: <conf-name>2023 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>; <year>2023 May 29&#x2013;Jun 2</year>; <publisher-loc>London, UK</publisher-loc>. p. <fpage>3456</fpage>&#x2013;<lpage>62</lpage>. doi:<pub-id pub-id-type="doi">10.1109/ICRA48891.2023.10161500</pub-id>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Squeeze-and-excitation networks</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. <year>2020</year>;<volume>42</volume>(<issue>8</issue>):<fpage>2011</fpage>&#x2013;<lpage>23</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TPAMI.2019.2913372</pub-id>; <pub-id pub-id-type="pmid">31034408</pub-id></mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Li</surname> <given-names>P</given-names></string-name>, <string-name><surname>Zuo</surname> <given-names>W</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>Q</given-names></string-name></person-group>. <article-title>ECA-Net: efficient channel attention for deep convolutional neural networks</article-title>. In: <conf-name>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2020 Jun 13&#x2013;19</year>; <publisher-loc>Seattle, WA, USA</publisher-loc>. p. <fpage>11531</fpage>&#x2013;<lpage>9</lpage>. [cited 2025 Jun 4]. Available from: <ext-link ext-link-type="uri" xlink:href="https://openaccess.thecvf.com/content_CVPR_2020/papers/Wang_ECA-Net_Efficient_Channel_Attention_for_Deep_Convolutional_Neural_Networks_CVPR_2020_paper.pdf">https://openaccess.thecvf.com/content_CVPR_2020/papers/Wang_ECA-Net_Efficient_Channel_Attention_for_Deep_Convolutional_Neural_Networks_CVPR_2020_paper.pdf</ext-link>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Hou</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>D</given-names></string-name>, <string-name><surname>Feng</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Coordinate attention for efficient mobile network design</article-title>. In: <conf-name>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2021 Jun 20&#x2013;25</year>; <publisher-loc>Nashville, TN, USA</publisher-loc>. p. <fpage>13713</fpage>&#x2013;<lpage>22</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01351</pub-id>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Woo</surname> <given-names>S</given-names></string-name>, <string-name><surname>Park</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lee</surname> <given-names>JY</given-names></string-name>, <string-name><surname>Kweon</surname> <given-names>IS</given-names></string-name></person-group>. <article-title>CBAM: convolutional block attention module</article-title>. In: <conf-name>European Conference on Computer Vision (ECCV)</conf-name>; <year>2018 Sep 8&#x2013;14</year>; <publisher-loc>Munich, Germany</publisher-loc>. p. <fpage>3</fpage>&#x2013;<lpage>19</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-030-01234-2_1</pub-id>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kelenyi</surname> <given-names>B</given-names></string-name>, <string-name><surname>Domsa</surname> <given-names>V</given-names></string-name>, <string-name><surname>Tamas</surname> <given-names>L</given-names></string-name></person-group>. <article-title>SAM-Net: self-attention based feature matching with spatial transformers and knowledge distillation</article-title>. <source>Expert Syst Appl</source>. <year>2024</year>;<volume>242</volume>(<issue>2</issue>):<fpage>122804</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eswa.2023.122804</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Xia</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>X</given-names></string-name>, <string-name><surname>Song</surname> <given-names>S</given-names></string-name>, <string-name><surname>Li</surname> <given-names>LE</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Vision transformer with deformable attention</article-title>. <comment>arXiv:2201.00520. 2022</comment>. </mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Li</surname> <given-names>L</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Weng</surname> <given-names>K</given-names></string-name>, <string-name><surname>Geng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>YOLOv6: a single-stage object detection framework for industrial applications</article-title>. <comment>arXiv:2209.02976. 2022</comment>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>CY</given-names></string-name>, <string-name><surname>Bochkovskiy</surname> <given-names>A</given-names></string-name>, <string-name><surname>Liao</surname> <given-names>HYM</given-names></string-name></person-group>. <article-title>YOLOv7: trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</article-title>. In: <conf-name>Proceedings of the 2023 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2023 Jun 17&#x2013;24</year>; <publisher-loc>Vancouver, BC, Canada</publisher-loc>. p. <fpage>7464</fpage>&#x2013;<lpage>75</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR52729.2023.00721</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Reis</surname> <given-names>D</given-names></string-name>, <string-name><surname>Kupec</surname> <given-names>J</given-names></string-name>, <string-name><surname>Hong</surname> <given-names>J</given-names></string-name>, <string-name><surname>Daoudi</surname> <given-names>A</given-names></string-name></person-group>. <article-title>Real-time flying object detection with YOLOv8</article-title>. <comment>arXiv:2305.09972. 2024</comment>. doi:<pub-id pub-id-type="doi">10.48550/arXiv.2305.09972</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Fang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liao</surname> <given-names>B</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Fang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Qi</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>R</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>You only look at one sequence: rethinking transformer in vision through object detection</article-title>. <comment>arXiv:2106.00666. 2021</comment>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhao</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Lv</surname> <given-names>W</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>J</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>G</given-names></string-name>, <string-name><surname>Dang</surname> <given-names>Q</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>DETRs beat YOLOs on real-time object detection</article-title>. <comment>arXiv:2304.08069. 2024</comment>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>C</given-names></string-name>, <string-name><surname>Xiao</surname> <given-names>B</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>C</given-names></string-name>, <string-name><surname>Yuan</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Sang</surname> <given-names>N</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Lite-HRNet: a lightweight high-resolution network</article-title>. In: <conf-name>Proceedings of the 2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2021 Jun 20&#x2013;25</year>; <publisher-loc>Nashville, TN, USA</publisher-loc>. p. <fpage>10440</fpage>&#x2013;<lpage>50</lpage>. [cited 2025 Jun 4]. Available from: <ext-link ext-link-type="uri" xlink:href="https://openaccess.thecvf.com/content/CVPR2021/html/Yu_Lite-HRNet_A_Lightweight_High-Resolution_Network_CVPR_2021_paper.html">https://openaccess.thecvf.com/content/CVPR2021/html/Yu_Lite-HRNet_A_Lightweight_High-Resolution_Network_CVPR_2021_paper.html</ext-link>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Ren</surname> <given-names>S</given-names></string-name>, <string-name><surname>He</surname> <given-names>K</given-names></string-name>, <string-name><surname>Girshick</surname> <given-names>R</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>J</given-names></string-name></person-group>. <chapter-title>Faster R-CNN: towards real-time object detection with region proposal networks</chapter-title>. Vol. 28, In: <source>Advances in Neural Information Processing Systems (NeurIPS);</source> 2015. [cited 2025 Jun 4]. Available from: <ext-link ext-link-type="uri" xlink:href="https://papers.nips.cc/paper_files/paper/2015/hash/14bfa6bb14875e45bba028a21ed38046-Abstract.html">https://papers.nips.cc/paper_files/paper/2015/hash/14bfa6bb14875e45bba028a21ed38046-Abstract.html</ext-link>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>W</given-names></string-name>, <string-name><surname>Anguelov</surname> <given-names>D</given-names></string-name>, <string-name><surname>Erhan</surname> <given-names>D</given-names></string-name>, <string-name><surname>Szegedy</surname> <given-names>C</given-names></string-name>, <string-name><surname>Reed</surname> <given-names>S</given-names></string-name>, <string-name><surname>Fu</surname> <given-names>CY</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>SSD: single shot MultiBox detector</article-title>. In: <conf-name>Proceedings of the European Conference on Computer Vision (ECCV). Vol. 9905.</conf-name> <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>; 2016. p. <fpage>21</fpage>&#x2013;<lpage>37</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-319-46448-0_2</pub-id>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>GraphCrack: graph-based multi-modal attention network for road crack detection</article-title>. <source>Neural Networks</source>. <year>2023</year>;<volume>160</volume>(<issue>1</issue>):<fpage>87</fpage>&#x2013;<lpage>97</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.neunet.2023.02.016</pub-id>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>B</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>G</given-names></string-name></person-group>. <article-title>SODD-YOLOv8: an insulator defect detection algorithm based on feature enhancement and variable row convolution</article-title>. <source>Meas Sci Technol</source>. <year>2024</year>;<volume>36</volume>(<issue>1</issue>):<fpage>015401</fpage>. doi:<pub-id pub-id-type="doi">10.1088/1361-6501/ad824f</pub-id>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Howard</surname> <given-names>AG</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>B</given-names></string-name>, <string-name><surname>Kalenichenko</surname> <given-names>D</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Weyand</surname> <given-names>T</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>MobileNets: efficient convolutional neural networks for mobile vision applications</article-title>. <comment>arXiv:1704.04861. 2017</comment>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yuan</surname> <given-names>G</given-names></string-name>, <string-name><surname>Wen</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Evangelidis</surname> <given-names>G</given-names></string-name>, <string-name><surname>Tulyakov</surname> <given-names>S</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>EfficientFormer: vision transformers at MobileNet speed</article-title>. In: <conf-name>Advances in Neural Information Processing Systems (NeurIPS); 2022 Nov 28; New Orleans, LA, USA</conf-name>. p. <fpage>12934</fpage>&#x2013;<lpage>49</lpage>. [cited 2025 Jun 4]. Available from: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2022/file/5452ad8ee6ea6e7dc41db1cbd31ba0b8-Paper-Conference.pdf">https://proceedings.neurips.cc/paper_files/paper/2022/file/5452ad8ee6ea6e7dc41db1cbd31ba0b8-Paper-Conference.pdf</ext-link>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Sandler</surname> <given-names>M</given-names></string-name>, <string-name><surname>Howard</surname> <given-names>A</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhmoginov</surname> <given-names>A</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>L</given-names></string-name></person-group>. <article-title>MobileNetV2: inverted residuals and linear bottlenecks</article-title>. In: <conf-name>IEEE Conference on Computer Vision and Pattern Recognition (CVPR); 2018 Jun 18&#x2013;23; Salt Lake City, UT, USA</conf-name>. p. <fpage>4510</fpage>&#x2013;<lpage>20</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2018.00474</pub-id>.</mixed-citation></ref>
<ref id="ref-34"><label>[34]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ding</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Han</surname> <given-names>J</given-names></string-name>, <string-name><surname>Ding</surname> <given-names>G</given-names></string-name></person-group>. <article-title>RepVGG: making VGG-style convnets great again</article-title>. In: <conf-name>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2021 Jun 20&#x2013;25</year>; <publisher-loc>Nashville, TN, USA</publisher-loc>. p. <fpage>13733</fpage>&#x2013;<lpage>42</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01354</pub-id>.</mixed-citation></ref>
<ref id="ref-35"><label>[35]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Cheng</surname> <given-names>P</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Cong</surname> <given-names>W</given-names></string-name>, <string-name><surname>Zang</surname> <given-names>C</given-names></string-name></person-group>. <chapter-title>Tiny-YOLOv7: tiny object detection model for drone imagery</chapter-title>. In: <person-group person-group-type="editor"><string-name><surname>Lu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Ouyang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>H</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Dong</surname> <given-names>J</given-names></string-name> <etal>et al.</etal></person-group>, editors. <source>Image and graphics</source>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name>; <year>2023</year>. p. <fpage>53</fpage>&#x2013;<lpage>65</lpage>. doi:<pub-id pub-id-type="doi">10.1007/978-3-031-46311-2_5</pub-id>.</mixed-citation></ref>
<ref id="ref-36"><label>[36]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>F</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Du</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>X</given-names></string-name></person-group>. <article-title>Improving YOLOv7-tiny for infrared and visible light image object detection on drones</article-title>. <source>Remote Sens</source>. <year>2023</year>;<volume>15</volume>(<issue>13</issue>):<fpage>3214</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs15133214</pub-id>.</mixed-citation></ref>
<ref id="ref-37"><label>[37]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Terven</surname> <given-names>J</given-names></string-name>, <string-name><surname>C&#x00F3;rdova-Esparza</surname> <given-names>DM</given-names></string-name>, <string-name><surname>Romero-Gonz&#x00E1;lez</surname> <given-names>JA</given-names></string-name></person-group>. <article-title>A comprehensive review of YOLO architectures in computer vision: from YOLOv1 to YOLOv8 and YOLO-NAS</article-title>. <source>Mach Learn Knowl Extr</source>. <year>2023</year>;<volume>5</volume>(<issue>4</issue>):<fpage>1680</fpage>&#x2013;<lpage>716</lpage>. doi:<pub-id pub-id-type="doi">10.3390/make5040083</pub-id>.</mixed-citation></ref>
<ref id="ref-38"><label>[38]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Lin</surname> <given-names>TY</given-names></string-name>, <string-name><surname>Doll&#x00E1;r</surname> <given-names>P</given-names></string-name>, <string-name><surname>Girshick</surname> <given-names>R</given-names></string-name>, <string-name><surname>He</surname> <given-names>K</given-names></string-name>, <string-name><surname>Hariharan</surname> <given-names>B</given-names></string-name>, <string-name><surname>Belongie</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Feature pyramid networks for object detection</article-title>. In: <conf-name>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2017 Jul 21&#x2013;26</year>; <publisher-loc>Honolulu, HI, USA</publisher-loc>. p. <fpage>2117</fpage>&#x2013;<lpage>25</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2017.106</pub-id>.</mixed-citation></ref>
<ref id="ref-39"><label>[39]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>S</given-names></string-name>, <string-name><surname>Qi</surname> <given-names>L</given-names></string-name>, <string-name><surname>Qin</surname> <given-names>H</given-names></string-name>, <string-name><surname>Shi</surname> <given-names>J</given-names></string-name>, <string-name><surname>Jia</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Path aggregation network for instance segmentation</article-title>. In: <conf-name>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2018 Jun 18&#x2013;23</year>; <publisher-loc>Salt Lake City, UT, USA</publisher-loc>. p. <fpage>8759</fpage>&#x2013;<lpage>68</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR.2018.00913</pub-id>.</mixed-citation></ref>
<ref id="ref-40"><label>[40]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Tan</surname> <given-names>M</given-names></string-name>, <string-name><surname>Pang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Le</surname> <given-names>QV</given-names></string-name></person-group>. <article-title>EfficientDet: scalable and efficient object detection</article-title>. In: <conf-name>IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2020 Jun 13&#x2013;19</year>; <publisher-loc>Seattle, WA, USA</publisher-loc>. p. <fpage>10781</fpage>&#x2013;<lpage>90</lpage>. doi:<pub-id pub-id-type="doi">10.1109/CVPR42600.2020.01080</pub-id>.</mixed-citation></ref>
<ref id="ref-41"><label>[41]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>C</given-names></string-name></person-group>. <article-title>Attention-guided multi-granularity fusion model for video summarization</article-title>. <source>Expert Syst Appl</source>. <year>2024</year>;<volume>249</volume>(<issue>8</issue>):<fpage>123568</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eswa.2024.123568</pub-id>.</mixed-citation></ref>
<ref id="ref-42"><label>[42]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yao</surname> <given-names>F</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Ding</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhong</surname> <given-names>G</given-names></string-name>, <string-name><surname>Li</surname> <given-names>S</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>Attention-guided multi-scale fusion network for similar objects semantic segmentation</article-title>. <source>Cogn Comput</source>. <year>2024</year>;<volume>16</volume>(<issue>1</issue>):<fpage>366</fpage>&#x2013;<lpage>76</lpage>. doi:<pub-id pub-id-type="doi">10.1007/s12559-023-10206-8</pub-id>.</mixed-citation></ref>
<ref id="ref-43"><label>[43]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Tang</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>TAG-fusion: two-stage attention guided multi-modal fusion network for semantic segmentation</article-title>. <source>Digit Signal Process</source>. <year>2025</year>;<volume>156</volume>(<issue>4</issue>):<fpage>104807</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.dsp.2024.104807</pub-id>.</mixed-citation></ref>
<ref id="ref-44"><label>[44]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Lu</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Multiscale channel attention-driven graph dynamic fusion learning method for robust fault diagnosis</article-title>. <source>IEEE Trans Ind Inform</source>. <year>2024</year>;<volume>20</volume>(<issue>9</issue>):<fpage>11002</fpage>&#x2013;<lpage>13</lpage>. doi:<pub-id pub-id-type="doi">10.1109/TII.2024.3397401</pub-id>.</mixed-citation></ref>
<ref id="ref-45"><label>[45]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Xu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Xiong</surname> <given-names>D</given-names></string-name>, <string-name><surname>van Genabith</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Q</given-names></string-name></person-group>. <article-title>Efficient context-aware neural machine translation with layer-wise weighting and input-aware gating</article-title>. In: <conf-name>Proceedings of the 29th International Joint Conference on Artificial Intelligence (IJCAI); 2021 Jan 7&#x2013;15; Yokohama, Japan</conf-name>. p. <fpage>3933</fpage>&#x2013;<lpage>40</lpage>. [cited 2025 Jun 4]. Available from: <ext-link ext-link-type="uri" xlink:href="https://api.semanticscholar.org/CorpusID:220483453">https://api.semanticscholar.org/CorpusID:220483453</ext-link>.</mixed-citation></ref>
<ref id="ref-46"><label>[46]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Shi</surname> <given-names>W</given-names></string-name>, <string-name><surname>Han</surname> <given-names>X</given-names></string-name>, <string-name><surname>Lewis</surname> <given-names>M</given-names></string-name>, <string-name><surname>Tsvetkov</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zettlemoyer</surname> <given-names>L</given-names></string-name>, <string-name><surname>Yih</surname> <given-names>W</given-names></string-name></person-group>. <article-title>Trusting your evidence: hallucinate less with context-aware decoding</article-title>. In: <conf-name>Proceedings of the 2024 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Mexico City, Mexico</conf-name>. <year>2024</year>. p. <fpage>783</fpage>&#x2013;<lpage>91</lpage>. doi:<pub-id pub-id-type="doi">10.18653/v1/2024.naacl-short.69</pub-id>.</mixed-citation></ref>
<ref id="ref-47"><label>[47]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Xie</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>X</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>L</given-names></string-name>, <string-name><surname>Shao</surname> <given-names>F</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>W</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>CANet: context aware network with dual-stream pyramid for medical image segmentation</article-title>. <source>Biomed Signal Process Control</source>. <year>2023</year>;<volume>81</volume>(<issue>1</issue>):<fpage>104437</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.bspc.2022.104437</pub-id>.</mixed-citation></ref>
<ref id="ref-48"><label>[48]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Mo</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>F</given-names></string-name>, <string-name><surname>Liao</surname> <given-names>Y</given-names></string-name></person-group>. <article-title>Review the state-of-the-art technologies of semantic segmentation based on deep learning</article-title>. <source>Neurocomputing</source>. <year>2022</year>;<volume>493</volume>:<fpage>626</fpage>&#x2013;<lpage>46</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.neucom.2022.01.005</pub-id>.</mixed-citation></ref>
<ref id="ref-49"><label>[49]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhong</surname> <given-names>K</given-names></string-name>, <string-name><surname>Jackson</surname> <given-names>T</given-names></string-name>, <string-name><surname>West</surname> <given-names>A</given-names></string-name>, <string-name><surname>Cosma</surname> <given-names>G</given-names></string-name></person-group>. <article-title>Natural language processing approaches in industrial maintenance: a systematic literature review</article-title>. <source>Procedia Comput Sci</source>. <year>2024</year>;<volume>232</volume>(<issue>4</issue>):<fpage>2082</fpage>&#x2013;<lpage>97</lpage>. doi:<pub-id pub-id-type="doi">10.1016/j.procs.2024.02.029</pub-id>.</mixed-citation></ref>
<ref id="ref-50"><label>[50]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Nam</surname> <given-names>W</given-names></string-name>, <string-name><surname>Jang</surname> <given-names>B</given-names></string-name></person-group>. <article-title>A survey on multimodal bidirectional machine learning translation of image and natural language processing</article-title>. <source>Expert Syst Appl</source>. <year>2024</year>;<volume>235</volume>(<issue>4</issue>):<fpage>121168</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.eswa.2023.121168</pub-id>.</mixed-citation></ref>
<ref id="ref-51"><label>[51]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Girshick</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Fast R-CNN</article-title>. In: <conf-name>Proceedings of the 2015 IEEE International Conference on Computer Vision (ICCV)</conf-name>; <year>2015 Dec 7&#x2013;13</year>; <publisher-loc>Santiago, Chile</publisher-loc>. p. <fpage>1440</fpage>&#x2013;<lpage>8</lpage>. [cited 2025 Jun 4]. Available from: <ext-link ext-link-type="uri" xlink:href="https://openaccess.thecvf.com/content_iccv_2015/html/Girshick_Fast_R-CNN_ICCV_2015_paper.html">https://openaccess.thecvf.com/content_iccv_2015/html/Girshick_Fast_R-CNN_ICCV_2015_paper.html</ext-link>.</mixed-citation></ref>
<ref id="ref-52"><label>[52]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>He</surname> <given-names>K</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>S</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Deep residual learning for image recognition</article-title>. In: <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>; <year>2016 Jun 27&#x2013;30</year>; <publisher-loc>Las Vegas, NV, USA</publisher-loc>. p. <fpage>770</fpage>&#x2013;<lpage>8</lpage>. [cited 2025 Jun 4]. Available from: <ext-link ext-link-type="uri" xlink:href="https://openaccess.thecvf.com/content_cvpr_2016/html/He_Deep_Residual_Learning_CVPR_2016_paper.html">https://openaccess.thecvf.com/content_cvpr_2016/html/He_Deep_Residual_Learning_CVPR_2016_paper.html</ext-link>.</mixed-citation></ref>
<ref id="ref-53"><label>[53]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dang</surname> <given-names>KB</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>CQ</given-names></string-name>, <string-name><surname>Tran</surname> <given-names>QC</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>H</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>TT</given-names></string-name>, <string-name><surname>Nguyen</surname> <given-names>DA</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Comparison between U-shaped structural deep learning models to detect landslide traces</article-title>. <source>Sci Total Environ</source>. <year>2024</year>;<volume>912</volume>:<fpage>169113</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.scitotenv.2023.169113</pub-id>; <pub-id pub-id-type="pmid">38065499</pub-id></mixed-citation></ref>
<ref id="ref-54"><label>[54]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>B</given-names></string-name>, <string-name><surname>Deng</surname> <given-names>F</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>P</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Han</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>WiTUnet: a U-shaped architecture integrating CNN and Transformer for improved feature alignment and local information fusion</article-title>. <source>Sci Rep</source>. <year>2024</year>;<volume>14</volume>(<issue>1</issue>):<fpage>25525</fpage>. doi:<pub-id pub-id-type="doi">10.1038/s41598-024-76886-w</pub-id>; <pub-id pub-id-type="pmid">39462127</pub-id></mixed-citation></ref>
<ref id="ref-55"><label>[55]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhou</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Bai</surname> <given-names>X</given-names></string-name>, <string-name><surname>Li</surname> <given-names>C</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Peng</surname> <given-names>G</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Semantic segmentation of surface cracks in urban comprehensive pipe galleries based on global attention</article-title>. <source>Sensors</source>. <year>2024</year>;<volume>24</volume>(<issue>3</issue>):<fpage>1005</fpage>. doi:<pub-id pub-id-type="doi">10.3390/s24031005</pub-id>; <pub-id pub-id-type="pmid">38339722</pub-id></mixed-citation></ref>
<ref id="ref-56"><label>[56]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liu</surname> <given-names>J</given-names></string-name>, <string-name><surname>Mao</surname> <given-names>S</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>L</given-names></string-name></person-group>. <article-title>Attention-based two-branch hybrid fusion network for medical image segmentation</article-title>. <source>Appl Sci</source>. <year>2024</year>;<volume>14</volume>(<issue>10</issue>):<fpage>4073</fpage>. doi:<pub-id pub-id-type="doi">10.3390/app14104073</pub-id>.</mixed-citation></ref>
<ref id="ref-57"><label>[57]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>CY</given-names></string-name>, <string-name><surname>Yeh</surname> <given-names>IH</given-names></string-name>, <string-name><surname>Liao</surname> <given-names>HYM</given-names></string-name></person-group>. <article-title>YOLOv9: learning what you want to learn using programmable gradient information</article-title>. <comment>arXiv:2402.13616. 2024</comment>.</mixed-citation></ref>
<ref id="ref-58"><label>[58]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>A</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>H</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>K</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Han</surname> <given-names>J</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>YOLOv10: real-time end-to-end object detection</article-title>. <comment>arXiv:2405.14458. 2024</comment>.</mixed-citation></ref>
<ref id="ref-59"><label>[59]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Xiao</surname> <given-names>X</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Xie</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>H</given-names></string-name>, <string-name><surname>Roy</surname> <given-names>SK</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>TD-RD: a top-down benchmark with real-time framework for road damage detection</article-title>. <comment>arXiv:2501.14302. 2025.</comment></mixed-citation></ref>
<ref id="ref-60"><label>[60]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>G</given-names></string-name>, <string-name><surname>Chang</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Lv</surname> <given-names>W</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>C</given-names></string-name>, <string-name><surname>Cui</surname> <given-names>C</given-names></string-name>, <string-name><surname>Ji</surname> <given-names>W</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>PP-PicoDet: a better real-time object detector on mobile devices</article-title>. <comment>arXiv:2111.00902. 2021</comment>.</mixed-citation></ref>
</ref-list>
</back></article>