<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">80341</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2026.080341</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>MFCI-YOLO: Lightweight UAV Aerial Photography Small Object Detection Method Based on Multi-Scale Feature Fusion and Contextual Information</article-title>
<alt-title alt-title-type="left-running-head">MFCI-YOLO: Lightweight UAV Aerial Photography Small Object Detection Method Based on Multi-Scale Feature Fusion and Contextual Information</alt-title>
<alt-title alt-title-type="right-running-head">MFCI-YOLO: Lightweight UAV Aerial Photography Small Object Detection Method Based on Multi-Scale Feature Fusion and Contextual Information</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Wang</surname><given-names>Weiguang</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Li</surname><given-names>Jincai</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Liu</surname><given-names>Mengqi</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Liu</surname><given-names>Mengke</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Zhang</surname><given-names>Yuan</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-6" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Wu</surname><given-names>Jingyan</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><email>wujingyan@jxust.edu.cn</email></contrib>
<contrib id="author-7" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Liu</surname><given-names>Yang</given-names></name><xref ref-type="aff" rid="aff-3">3</xref><email>liuyang@jxust.edu.cn</email></contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western"><surname>Lou</surname><given-names>Junbin</given-names></name><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-9" contrib-type="author">
<name name-style="western"><surname>He</surname><given-names>Yixin</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<aff id="aff-1"><label>1</label><institution>School of Information Engineering, Henan University of Science and Technology</institution>, <addr-line>Luoyang</addr-line>, <country>China</country></aff>
<aff id="aff-2"><label>2</label><institution>Industry Research Institute of Intelligent Systems, Longmen Laboratory</institution>, <addr-line>Luoyang</addr-line>, <country>China</country></aff>
<aff id="aff-3"><label>3</label><institution>School of Electrical and Information Engineering, Guangdong Baiyun University</institution>, <addr-line>Guangzhou</addr-line>, <country>China</country></aff>
<aff id="aff-4"><label>4</label><institution>College of Mechanical Engineering, Jiaxing University</institution>, <addr-line>Jiaxing</addr-line>, <country>China</country></aff>
<aff id="aff-5"><label>5</label><institution>College of Information Science and Engineering, Jiaxing University</institution>, <addr-line>Jiaxing</addr-line>, <country>China</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Authors: Jingyan Wu. Email: <email>wujingyan@jxust.edu.cn</email>; Yang Liu. Email: <email>liuyang@jxust.edu.cn</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2026</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>15</day><month>06</month><year>2026</year>
</pub-date>
<volume>88</volume>
<issue>2</issue>
<elocation-id>82</elocation-id>
<history>
<date date-type="received">
<day>07</day>
<month>02</month>
<year>2026</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>05</month>
<year>2026</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2026 The Authors. Published by Tech Science Press.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder>The Authors</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_80341.pdf"></self-uri>
<abstract>
<p>To improve the accuracy of small object feature detection in complex backgrounds for Unmanned Aerial Vehicle (UAV) aerial photography and reduce computational complexity, we propose the lightweight UAV aerial photography small object detection method based on multi-scale feature fusion and contextual information. Firstly, by introducing the grouped content-aware reassembly (GCA) operator and designing lightweight pinwheel context convolution (LPConv), we extend the feature fusion path to the P2 layer, constructing a lightweight multi-scale feature fusion network (SG-PANet). Through the decoupling of fine-grained small object features and background interference features by the GCA operator, combined with the anisotropic receptive field constructed by LPConv, our proposed method can effectively preserve the geometric details of small objects. Furthermore, we introduce the cross-stage dense feature refinement (CSPStage) module as the pre-refining unit of the detection head, and use the full history state awareness mechanism to strengthen feature reuse and gradient propagation to solve the problem of feature degradation across layers. We utilize the Wise-IoU v3 loss function to dynamically optimize the gradient gains of high-quality and low-quality samples, thereby enhancing the detection accuracy and convergence speed of the proposed method in complex scenarios. Finally, we verified the superiority and generalization of the proposed method on the VisDrone2019 dataset and DOTAv1.5 dataset. The results show that compared with YOLOv11n, MFCI-YOLO&#x2019;s detection mAP50-95 increased by 11.1%, small object mAP50 increased by 16.1%, and mAP50 reached 80.3%. It provides a practical solution for detecting small objects in dense scenes.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Small object detection</kwd>
<kwd>UAV aerial photography</kwd>
<kwd>YOLOv11n</kwd>
<kwd>grouped structure</kwd>
<kwd>context-aware</kwd>
<kwd>feature fusion</kwd>
<kwd>lightweight</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>Natural Science Foundation of Henan Province</funding-source>
<award-id>252300423317</award-id>
</award-group>
<award-group id="awg2">
<funding-source>Science and Technology Research Project of Henan Province</funding-source>
<award-id>262102211081</award-id>
</award-group>
<award-group id="awg3">
<funding-source>Key Scientific Research Projects of Colleges and Universities in Henan Province</funding-source>
<award-id>25B510012</award-id>
</award-group>
<award-group id="awg4">
<funding-source>R&#x0026;D Program of Zhejiang</funding-source>
<award-id>2026LDC01003(JT)</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>With UAVs widely deployed in the global low-altitude economy, establishing robust communication and edge-computing infrastructures&#x2014;such as Internet of Things (IoT)-enhanced emergency communications [<xref ref-type="bibr" rid="ref-1">1</xref>], Multiple-Input Multiple-Output (MIMO) cooperative networks [<xref ref-type="bibr" rid="ref-2">2</xref>], Non-Orthogonal Multiple Access with Mobile Edge Computing (NOMA-MEC) offloading [<xref ref-type="bibr" rid="ref-3">3</xref>], and Reconfigurable Intelligent Surface (RIS)-based Integrated Sensing and Communication (ISAC) frameworks [<xref ref-type="bibr" rid="ref-4">4</xref>]&#x2014;has become a foundational prerequisite. However, the ultimate efficacy of these advanced networks relies entirely on accurate visual perception. In such high-altitude missions, critical targets inherently appear as extremely small objects. Consequently, robust small-object detection has become the perceptual bottleneck and crucial enabler for UAV intelligence within these complex networks. However, existing UAV aerial detection suffers from severe missed or false detections due to extreme perspective changes and complex scenes.</p>
<p>Selecting appropriate algorithms is key for accurate scene analysis. Mainstream Convolutional Neural Network (CNN)-based object detection algorithms perform poorly in UAV aerial small-object detection, as UAV aerial photography has three features: (1) small object pixel ratio; (2) limited hardware computing power and storage; (3) complex scenes with strong external interference.</p>
<p>YOLO series models are widely used in UAV aerial detection for their speed and accuracy. Li et al. [<xref ref-type="bibr" rid="ref-5">5</xref>] proposed SOD-YOLO (enhanced YOLOv8) to solve blurred small-object features, but its high-resolution head limits edge inference speed. Wan et al. [<xref ref-type="bibr" rid="ref-6">6</xref>] proposed a dynamic attention-based ultra-lightweight method to reduce information loss, yet complex attention impairs real-time performance. Yu and Mo [<xref ref-type="bibr" rid="ref-7">7</xref>] proposed YOLO-GCOF to alleviate computational bottlenecks, but lightweight pruning reduces robustness. In summary, UAV object detection research focuses on: (1) improving accuracy while maintaining efficiency via lightweight design; (2) enhancing small-object feature representation in complex backgrounds.</p>
<p>Based on YOLOv11n, we propose a lightweight UAV aerial small-object detection method using multi-scale feature fusion and contextual information, addressing feature drown-out, misalignment and limited localization robustness in extremely small-object cross-scale fusion. Specifically, to meet the strict deployment standards of resource-constrained UAV edge devices, we explicitly set our lightweight design objects as: total parameters <inline-formula id="ieqn-1"><mml:math id="mml-ieqn-1"><mml:mrow><mml:mo>&#x003C;</mml:mo></mml:mrow><mml:mn>5</mml:mn></mml:math></inline-formula>M and computational complexity <inline-formula id="ieqn-2"><mml:math id="mml-ieqn-2"><mml:mrow><mml:mo>&#x003C;</mml:mo></mml:mrow><mml:mn>50</mml:mn></mml:math></inline-formula> Giga Floating-point Operations Per Second (GFLOPs), while ensuring real-time inference capability. Unlike recently proposed YOLO-based UAV detectors cited in <xref ref-type="sec" rid="s2">Section 2</xref>&#x2014;such as SOD-YOLO [<xref ref-type="bibr" rid="ref-5">5</xref>], which relies on computationally heavy high-resolution heads, or DAU-YOLO [<xref ref-type="bibr" rid="ref-6">6</xref>], which utilizes complex attention mechanisms that remain susceptible to feature drowning in dense backgrounds&#x2014;our method fundamentally resolves the efficiency-accuracy conflict. We advance beyond these works by introducing extremely lightweight orthogonal convolutions and decoupled reconstruction mechanisms rather than stacking redundant parameters.</p>
<p>To provide a sharper distinction between our architectural improvements and novel modules, the core innovations and contributions of this paper are summarized as follows:</p>
<p>(1) Novel Modular Operators (LPConv &#x0026; GCA): We propose the Lightweight Pinwheel Context Convolution (LPConv) to capture anisotropic contextual features of small objects via a dual-stream orthogonal alignment mechanism, drastically reducing parameter overhead. Concurrently, we introduce the Grouped Content-Aware (GCA) operator, a novel decoupled upsampling module that isolates fine-grained object features from complex backgrounds to fundamentally prevent feature drowning.</p>
<p>(2) Architectural Improvement (SG-PANet): Leveraging the aforementioned modules, we design a structural paradigm shift named the Scale-Gradient and Context-Aware Feature Fusion Network (SG-PANet). By abandoning the deep P5 layer and extending the fusion hierarchy directly to the high-resolution P2 layer, this architecture explicitly preserves the geometric details of extremely small objects that are typically lost in standard YOLO frameworks.</p>
<p>(3) Feature Refinement and Dynamic Optimization: We integrate the CSPStage module as a pre-refining unit before the detection head to leverage full-history state awareness, bridging semantic gaps and mitigating deep feature decay. Additionally, the Wise-IoU v3 loss function is employed to dynamically allocate gradient gains, significantly boosting localization robustness for low-quality small object instances.</p>
<p>(4) MFCI-YOLO balances accuracy and efficiency, achieving 48.2% mAP50 at 97 Frames Per Second (FPS) on VisDrone2019. An 80.3% mAP50 on DOTAv1.5 further validates its generalization for practical small-object detection in dense scenes.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related Work</title>
<p>To resolve the core conflict between high-precision detection and computational constraints in UAV aerial photography, two main technical approaches are adopted: feature enhancement to strengthen effective features in small-object and occlusion scenarios, and lightweight network design to reduce overhead, balancing performance and deployability.</p>
<p>For small-object feature loss and severe background interference in UAV aerial photography, studies focus on feature fusion optimization. Jian et al. [<xref ref-type="bibr" rid="ref-8">8</xref>] introduced Bidirectional Feature Pyramid Network (BiFPN) into YOLOv5 to solve unidirectional fusion information loss. Ma and Wang [<xref ref-type="bibr" rid="ref-9">9</xref>] propose a dynamically weighted multi-scale fusion module to improve small-object recall. Lai et al. [<xref ref-type="bibr" rid="ref-10">10</xref>] design an efficient framework with SSFF and MSFE modules to alleviate dense-scene feature blurring. Qiao et al. [<xref ref-type="bibr" rid="ref-11">11</xref>] use recursive feature pyramids, and Sun et al. [<xref ref-type="bibr" rid="ref-12">12</xref>] adopt heterogeneous architectures to optimize feature representation.</p>
<p>To address edge deployment&#x2019;s computational constraints and low accuracy, researches focus on lightweight network reconstruction. Ye and Li [<xref ref-type="bibr" rid="ref-13">13</xref>] use Ghost Bottleneck to reduce complexity and improve performance. Ju et al. [<xref ref-type="bibr" rid="ref-14">14</xref>] integrate LSKA and DCNv4 into YOLO for better geometric perception. Zhao et al. [<xref ref-type="bibr" rid="ref-15">15</xref>] propose a lightweight scheme for UAV search-and-rescue to detect minute objects on low-power devices. Ji et al. [<xref ref-type="bibr" rid="ref-16">16</xref>] combine lightweight convolutions with attention, and Zhu et al. [<xref ref-type="bibr" rid="ref-17">17</xref>] designe a dynamic sparse attention mechanism to balance performance and efficiency.</p>
<p>Despite recent advancements, existing methods exhibit critical bottlenecks. BiFPN and Recursive FPN lack noise suppression, lightweight convolutions sacrifice contextual awareness, and traditional heads fail in dense occlusions. Recent advancements utilizing Spatial Pyramid Multi-scale Common Convolution (SPMCC) [<xref ref-type="bibr" rid="ref-18">18</xref>], Excitation and Modulation Attention (EMA) [<xref ref-type="bibr" rid="ref-19">19</xref>], and content-aware reassembly [<xref ref-type="bibr" rid="ref-20">20</xref>] improve saliency, yet their channel-shared operations often cause small-target &#x2018;feature drowning&#x2019;. Meanwhile, architectures utilizing composite multi-scale fusion [<xref ref-type="bibr" rid="ref-21">21</xref>], multi-stage path aggregation [<xref ref-type="bibr" rid="ref-22">22</xref>], and spatially enhanced polarity sensing [<xref ref-type="bibr" rid="ref-23">23</xref>] achieve high precision but incur prohibitive computational overhead for resource-constrained UAV deployment. To resolve these conflicts, we propose MFCI-YOLO. By synergizing GCA for explicit noise decoupling, LPConv for anisotropic context capture, and CSPStage for occlusion refinement, our framework significantly improves localization accuracy and false-alarm suppression while maintaining a strictly lightweight profile.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>An Improved Lightweight Multi-Scale Feature Fusion Method for Small Object Detection</title>
<sec id="s3_1">
<label>3.1</label>
<title>MFCI-YOLO Network Architecture</title>
<p>To mitigate small-object feature loss and background interference in UAV imagery, we propose a lightweight YOLOv11n-based detector. As shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>, our method integrates LPConv for orthogonal feature alignment and geometric detail capture via a dynamic channel strategy. SG-PANet constructs a high-resolution P2 interaction space, while the GCA operator suppresses cross-scale fusion noise, ensuring semantic fidelity for faint objects. It is important to note that MFCI-YOLO refers to our overall end-to-end detection framework, whereas the SG-PANet specifically designates the novel neck architecture (feature fusion network) embedded within it.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>MFCI-YOLO network architecture.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-1.tif"/>
</fig>
<p>In the refinement and detection stage, CSPStage is deployed before the head to denoise and reconstruct features via full-history state awareness, bridging semantic gaps and enhancing robustness in occlusions. Finally, the Wise-IoU v3 loss function prioritizes high-quality small object instances, significantly boosting performance while maintaining real-time efficiency.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Lightweight Pinwheel Context Convolution</title>
<p>UAV-detected objects show distinct anisotropic elongated features. Traditional 3 <inline-formula id="ieqn-3"><mml:math id="mml-ieqn-3"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 3 convolution with square receptive fields introduces excessive background noise when processing such features, so we propose a lightweight pinwheel context convolution (LPConv) based on orthogonal center alignment and dual-stream complementarity.</p>
<p>As shown in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>, the input feature tensor is defined as: <inline-formula id="ieqn-4"><mml:math id="mml-ieqn-4"><mml:mi>X</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>. Where <inline-formula id="ieqn-5"><mml:math id="mml-ieqn-5"><mml:mi>X</mml:mi></mml:math></inline-formula> is the input tensor, <inline-formula id="ieqn-6"><mml:math id="mml-ieqn-6"><mml:mi>C</mml:mi></mml:math></inline-formula> is channel count, and <inline-formula id="ieqn-7"><mml:math id="mml-ieqn-7"><mml:mi>H</mml:mi><mml:mo>,</mml:mo><mml:mi>W</mml:mi></mml:math></inline-formula> are its height and width.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Schematic diagram of LPConv.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-2.tif"/>
</fig>
<p>To improve detection and avoid channel dimension explosion, we split input features into a processing stream (<inline-formula id="ieqn-8"><mml:math id="mml-ieqn-8"><mml:msub><mml:mi>X</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:math></inline-formula>) for feature enhancement and an identity stream (<inline-formula id="ieqn-9"><mml:math id="mml-ieqn-9"><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula>) for original feature preservation: <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:msub><mml:mi>X</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mn>1</mml:mn><mml:mo>:</mml:mo><mml:mfrac><mml:mi>C</mml:mi><mml:mn>2</mml:mn></mml:mfrac><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo stretchy="false">]</mml:mo></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mspace width="1em" /><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mrow><mml:mo stretchy="false">[</mml:mo><mml:mfrac><mml:mi>C</mml:mi><mml:mn>2</mml:mn></mml:mfrac><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo>:</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo stretchy="false">]</mml:mo></mml:mrow></mml:msub></mml:math></inline-formula>. Subscripts denote channel slicing, with <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:msub><mml:mi>X</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msubsup><mml:mrow><mml:mi mathvariant="double-struck">R</mml:mi></mml:mrow><mml:mn>2</mml:mn><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>.</p>
<p>For <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:msub><mml:mi>X</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:math></inline-formula>, we build a Local Windmill Extraction Unit with three anisotropic convolution branches: 3 <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 3 depthwise separable convolution (core, high-frequency texture), 1 <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> <italic>k</italic> (horizontal context), and <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:mi>k</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> (vertical context). Specifically, the <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mi>k</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:mi>k</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> orthogonal branches are selected to construct a cross-shaped receptive field that tightly aligns with the anisotropic, slender morphologies of typical aerial objects (e.g., vehicles, pedestrians) from a top-down view, effectively avoiding the diagonal background noise introduced by standard <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:mi>k</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>k</mml:mi></mml:math></inline-formula> square filters. Residual additive aggregation unifies these features:<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:msub><mml:mi>Y</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mtext>core</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2295;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mtext>hor</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2295;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mtext>ver</mml:mtext></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:msub><mml:mi>Y</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:math></inline-formula> is the processing stream output, <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msub><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mrow><mml:mtext>core</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mrow><mml:mtext>hor</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mrow><mml:mtext>ver</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> are the three branches, and <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:mo>&#x2295;</mml:mo></mml:math></inline-formula> is element-wise addition.</p>
<p>When LPConv is specifically employed as a downsampling module to replace the standard Conv layer, a coordinated spatial reduction strategy is applied to both streams. The initial 1 <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 1 splitting convolution maintains a stride of 1. Subsequently, the parallel depthwise convolutions (the <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula>, <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mi>k</mml:mi></mml:math></inline-formula>, and <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:mi>k</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> branches) in the processing stream operate with a stride of 2 (<inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn></mml:math></inline-formula>) to halve the spatial dimensions while extracting multi-scale contextual features. Synchronously, a <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>2</mml:mn></mml:math></inline-formula> Max Pooling operation is applied to the identity stream (<inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula>). This pooling step is essential to strictly align the spatial dimensions of <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:msub><mml:mi>Y</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:msub><mml:mi>Y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> (both reduced to <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mi>H</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:math></inline-formula>) prior to the final channel concatenation. For standard feature extraction without downsampling, the stride remains <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> and the Max Pooling operation is bypassed.</p>
<p>Finally, we remap and unify these complementary subspaces by concatenating <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:msub><mml:mi>Y</mml:mi><mml:mi>p</mml:mi></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-34"><mml:math id="mml-ieqn-34"><mml:msub><mml:mi>Y</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:math></inline-formula> along the channel dimension. A <inline-formula id="ieqn-35"><mml:math id="mml-ieqn-35"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> pointwise convolution then aggregates features and adjusts channel depth. The final LPConv output, <inline-formula id="ieqn-36"><mml:math id="mml-ieqn-36"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula>, is expressed as: <inline-formula id="ieqn-37"><mml:math id="mml-ieqn-37"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">t</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">f</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">t</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>Y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. Where <inline-formula id="ieqn-38"><mml:math id="mml-ieqn-38"><mml:mtext>Concat</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> is channel-wise splicing and <inline-formula id="ieqn-39"><mml:math id="mml-ieqn-39"><mml:msub><mml:mrow><mml:mi>&#x2131;</mml:mi></mml:mrow><mml:mrow><mml:mtext>fuse</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> is 1 <inline-formula id="ieqn-40"><mml:math id="mml-ieqn-40"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 1 fusion convolution.</p>
<p>In simpler terms, equations mathematically describe a split-and-conquer strategy. By dividing the feature maps, LPConv ensures that one half actively captures the orthogonal shapes of small objects, while the other half preserves the original uncorrupted details, achieving a highly efficient balance between contextual learning and parameter reduction.</p>
<p>Unlike conventional strip convolution methods that simply alternate <inline-formula id="ieqn-41"><mml:math id="mml-ieqn-41"><mml:mn>1</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mi>k</mml:mi></mml:math></inline-formula> and <inline-formula id="ieqn-42"><mml:math id="mml-ieqn-42"><mml:mi>k</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula> kernels to expand the receptive field at the cost of losing central high-frequency details, LPConv is driven by a dual-stream orthogonal alignment motivation. By explicitly decoupling the feature space into a processing stream for anisotropic contextual capture and an identity stream for fine-grained texture preservation, LPConv fundamentally avoids the detail degradation typical in extremely small objects during deep feature extraction.</p>
<p>For lightweight evaluation, assume input/output channels <inline-formula id="ieqn-43"><mml:math id="mml-ieqn-43"><mml:mi>C</mml:mi></mml:math></inline-formula>, core kernel <inline-formula id="ieqn-44"><mml:math id="mml-ieqn-44"><mml:msub><mml:mi>k</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula>, and strip kernel <inline-formula id="ieqn-45"><mml:math id="mml-ieqn-45"><mml:msub><mml:mi>k</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>p</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>k</mml:mi></mml:math></inline-formula>. The standard <inline-formula id="ieqn-46"><mml:math id="mml-ieqn-46"><mml:mn>3</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula> convolution parameter count is <inline-formula id="ieqn-47"><mml:math id="mml-ieqn-47"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>9</mml:mn><mml:msup><mml:mi>C</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula>. Because LPConv performs spatial convolution on only half the channels, its total parameter count is <inline-formula id="ieqn-48"><mml:math id="mml-ieqn-48"><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>P</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mi>C</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mfrac><mml:mi>C</mml:mi><mml:mn>2</mml:mn></mml:mfrac><mml:mo stretchy="false">(</mml:mo><mml:mn>9</mml:mn><mml:mo>+</mml:mo><mml:mn>2</mml:mn><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. The parameter compression ratio is thus defined as <inline-formula id="ieqn-49"><mml:math id="mml-ieqn-49"><mml:msub><mml:mi>&#x03B7;</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>L</mml:mi><mml:mi>P</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. Setting <inline-formula id="ieqn-50"><mml:math id="mml-ieqn-50"><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mn>128</mml:mn></mml:math></inline-formula> and <inline-formula id="ieqn-51"><mml:math id="mml-ieqn-51"><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>7</mml:mn></mml:math></inline-formula>, we obtain <inline-formula id="ieqn-52"><mml:math id="mml-ieqn-52"><mml:msub><mml:mi>&#x03B7;</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2248;</mml:mo><mml:mn>12.1</mml:mn><mml:mi mathvariant="normal">&#x0025;</mml:mi></mml:math></inline-formula>. This theoretically demonstrates that LPConv expands the effective receptive field to <inline-formula id="ieqn-53"><mml:math id="mml-ieqn-53"><mml:mn>7</mml:mn><mml:mo>&#x00D7;</mml:mo><mml:mn>7</mml:mn></mml:math></inline-formula> while retaining robust feature extraction capabilities using only <inline-formula id="ieqn-54"><mml:math id="mml-ieqn-54"><mml:mn>12.1</mml:mn><mml:mi mathvariant="normal">&#x0025;</mml:mi></mml:math></inline-formula> of the parameters of a standard convolution.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Optimized Upsampling Operator</title>
<p>The CARAFE operator proposed by Wang et al. [<xref ref-type="bibr" rid="ref-24">24</xref>] employs a dynamic kernel generation mechanism, yet its channel sharing approach exhibits limitations in UAV aerial photography scenarios. As background information dominates the scene, the shared kernel often functions as a smoothing filter, leading to the misidentification of small object edges as noise and their subsequent suppression. This results in the phenomenon of feature drowning.</p>
<p>To address this issue, we propose a group-content-aware (GCA) reconstruction method based on low-level decoupling. As illustrated in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, we partition the channel space <inline-formula id="ieqn-55"><mml:math id="mml-ieqn-55"><mml:mi>C</mml:mi></mml:math></inline-formula> into <inline-formula id="ieqn-56"><mml:math id="mml-ieqn-56"><mml:mi>G</mml:mi></mml:math></inline-formula> independent semantic groups (<inline-formula id="ieqn-57"><mml:math id="mml-ieqn-57"><mml:msub><mml:mi>C</mml:mi><mml:mi>g</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>C</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>G</mml:mi></mml:math></inline-formula> for each group) and process them through two parallel branches. In our implementation, the number of groups <inline-formula id="ieqn-58"><mml:math id="mml-ieqn-58"><mml:mi>G</mml:mi></mml:math></inline-formula> is empirically set to 4. The choice of <inline-formula id="ieqn-59"><mml:math id="mml-ieqn-59"><mml:mi>G</mml:mi></mml:math></inline-formula> significantly impacts both detection performance and computational efficiency. A smaller <inline-formula id="ieqn-60"><mml:math id="mml-ieqn-60"><mml:mi>G</mml:mi></mml:math></inline-formula> (e.g., <inline-formula id="ieqn-61"><mml:math id="mml-ieqn-61"><mml:mi>G</mml:mi></mml:math></inline-formula> &#x003D; 1) regresses to a channel-shared mechanism, failing to decouple the background from the object and leading to feature drowning. Conversely, an excessively large <inline-formula id="ieqn-62"><mml:math id="mml-ieqn-62"><mml:mi>G</mml:mi></mml:math></inline-formula> forces the network to generate too many independent kernels, which not only drastically increases the memory access cost (MAC) and computational overhead but also disrupts the synergistic feature representation within semantic channel groups. Setting <inline-formula id="ieqn-63"><mml:math id="mml-ieqn-63"><mml:mi>G</mml:mi></mml:math></inline-formula> &#x003D; 4 effectively strikes an optimal balance, ensuring sufficiently fine-grained semantic decoupling for small objects while adhering to strict lightweight deployment constraints.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Schematic diagram of GCA.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-3.tif"/>
</fig>
<p>In the group kernel prediction branch, input features undergo 1 <inline-formula id="ieqn-64"><mml:math id="mml-ieqn-64"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 1 convolution compression before being processed through group convolutions (content encoder), which concurrently output <inline-formula id="ieqn-65"><mml:math id="mml-ieqn-65"><mml:mi>G</mml:mi></mml:math></inline-formula> sets of independent parameters. Following Pixel Shuffle and Softmax normalisation, this generates <inline-formula id="ieqn-66"><mml:math id="mml-ieqn-66"><mml:mi>G</mml:mi></mml:math></inline-formula> distinct re-organised kernel tensors <inline-formula id="ieqn-67"><mml:math id="mml-ieqn-67"><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mi>G</mml:mi></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula>.</p>
<p>Simultaneously, within the grouping and reorganisation branch, the input <inline-formula id="ieqn-68"><mml:math id="mml-ieqn-68"><mml:mi>X</mml:mi></mml:math></inline-formula> is decomposed into <inline-formula id="ieqn-69"><mml:math id="mml-ieqn-69"><mml:mi>G</mml:mi></mml:math></inline-formula> sub-streams <inline-formula id="ieqn-70"><mml:math id="mml-ieqn-70"><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>G</mml:mi></mml:msub><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula>. Each sub-stream undergoes grouped dot product reorganisation solely with its corresponding kernel <inline-formula id="ieqn-71"><mml:math id="mml-ieqn-71"><mml:msub><mml:mi>W</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:math></inline-formula>:<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:msup><mml:mi>l</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo>,</mml:mo><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msubsup><mml:mo>=</mml:mo><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:munderover><mml:munderover><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:munderover><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:msup><mml:mi>l</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>g</mml:mi></mml:msubsup><mml:mo>&#x22C5;</mml:mo><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>l</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo>+</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow><mml:mi>g</mml:mi></mml:msubsup></mml:math></disp-formula>where <inline-formula id="ieqn-72"><mml:math id="mml-ieqn-72"><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:msup><mml:mi>l</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup><mml:mo>,</mml:mo><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msubsup></mml:math></inline-formula> denotes the pixel at position <inline-formula id="ieqn-73"><mml:math id="mml-ieqn-73"><mml:msup><mml:mi>l</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msup></mml:math></inline-formula>, channel <inline-formula id="ieqn-74"><mml:math id="mml-ieqn-74"><mml:mi>c</mml:mi></mml:math></inline-formula>, and group <inline-formula id="ieqn-75"><mml:math id="mml-ieqn-75"><mml:mi>g</mml:mi></mml:math></inline-formula>. <inline-formula id="ieqn-76"><mml:math id="mml-ieqn-76"><mml:msub><mml:mi>W</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:math></inline-formula> is the reorganization kernel with radius <inline-formula id="ieqn-77"><mml:math id="mml-ieqn-77"><mml:mi>r</mml:mi></mml:math></inline-formula>, while <inline-formula id="ieqn-78"><mml:math id="mml-ieqn-78"><mml:mo stretchy="false">(</mml:mo><mml:mi>n</mml:mi><mml:mo>,</mml:mo><mml:mi>m</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> represents the local coordinate offset.</p>
<p>To restore channel capacity and spatial alignment, high-frequency details are reaggregated by concatenating all <inline-formula id="ieqn-79"><mml:math id="mml-ieqn-79"><mml:mi>G</mml:mi></mml:math></inline-formula> reconstructed subflows <inline-formula id="ieqn-80"><mml:math id="mml-ieqn-80"><mml:mo fence="false" stretchy="false">{</mml:mo><mml:msubsup><mml:mi>X</mml:mi><mml:mn>1</mml:mn><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mi>X</mml:mi><mml:mi>G</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msubsup><mml:mo fence="false" stretchy="false">}</mml:mo></mml:math></inline-formula> along the channel dimension. The final output <inline-formula id="ieqn-81"><mml:math id="mml-ieqn-81"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> is expressed as: <inline-formula id="ieqn-82"><mml:math id="mml-ieqn-82"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">u</mml:mi><mml:mi mathvariant="normal">t</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">t</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>X</mml:mi><mml:mn>1</mml:mn><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mi>X</mml:mi><mml:mn>2</mml:mn><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msubsup><mml:mi>X</mml:mi><mml:mi>G</mml:mi><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula>. Where <inline-formula id="ieqn-83"><mml:math id="mml-ieqn-83"><mml:mtext>Concat</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> denotes the concatenation operation along the channel dimension.</p>
<p>The recombined sub-streams are reshaped from <inline-formula id="ieqn-84"><mml:math id="mml-ieqn-84"><mml:mi>B</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>G</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>C</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>G</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x00D7;</mml:mo><mml:mi>o</mml:mi><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>o</mml:mi><mml:mi>W</mml:mi></mml:math></inline-formula> back to the standard <inline-formula id="ieqn-85"><mml:math id="mml-ieqn-85"><mml:mi>B</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:math></inline-formula> format. Unlike CARAFE&#x2019;s shared smoothing kernels, GCA assigns dedicated high-frequency kernels to independent semantic groups. This prevents small-object textures from being suppressed by dominant backgrounds, effectively alleviating feature drowning and enhancing discriminative details for detection.</p>
<p>Intuitively, a globally shared kernel acts as a single &#x201C;broad brush&#x201D; that inevitably blurs fine object details into the dominant background. In contrast, our grouped GCA mechanism assigns dedicated &#x201C;fine-tipped brushes&#x201D; exclusively for the high-frequency textures of small objects, explicitly decoupling them from noise and completely avoiding the feature drowning effect.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>CSPStage Module</title>
<p>To address the high-frequency feature decay of small objects in deep convolutional networks, we draw inspiration from the designs of GiraffeDet [<xref ref-type="bibr" rid="ref-25">25</xref>] and CSPNet [<xref ref-type="bibr" rid="ref-26">26</xref>] to propose the Cross-Stage Dense Feature Refinement module (CSPStage). This module enhances the model&#x2019;s representational capability for low-pixel-count objects through full-spectrum feature reuse and structural reparameterization.</p>
<p>As shown in <xref ref-type="fig" rid="fig-4">Fig. 4</xref>, CSPStage employs a dual-stream feature evolution system. The input tensor <inline-formula id="ieqn-86"><mml:math id="mml-ieqn-86"><mml:mi>X</mml:mi></mml:math></inline-formula> is projected onto the baseline subspace <inline-formula id="ieqn-87"><mml:math id="mml-ieqn-87"><mml:msub><mml:mi>Y</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:math></inline-formula> and the evolutionary subspace <inline-formula id="ieqn-88"><mml:math id="mml-ieqn-88"><mml:msubsup><mml:mi>Y</mml:mi><mml:mn>2</mml:mn><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula>.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Schematic diagram of CSPStage architecture.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-4.tif"/>
</fig>
<p>In the baseline branch, 1 <inline-formula id="ieqn-89"><mml:math id="mml-ieqn-89"><mml:mo>&#x00D7;</mml:mo></mml:math></inline-formula> 1 convolutions construct truncated gradient highways, enabling raw high-resolution spatial features to bypass deep nonlinear transformations and reach the fusion layer directly, which ensures precise pixel-level localization of small objects.</p>
<p>Within the evolutionary branch, the dense aggregation mechanism integrates the outputs <inline-formula id="ieqn-90"><mml:math id="mml-ieqn-90"><mml:msubsup><mml:mi>Y</mml:mi><mml:mn>2</mml:mn><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msubsup></mml:math></inline-formula> from <inline-formula id="ieqn-91"><mml:math id="mml-ieqn-91"><mml:mi>n</mml:mi></mml:math></inline-formula> cascaded units. This unit employs a RepConv operator, which leverages a multi-branch topology during training to capture multi-scale features. During inference, it fuses equivalently into a single kernel, enhancing extraction robustness without increasing.</p>
<p>By fusing shallow spatial and deep semantic features, <inline-formula id="ieqn-92"><mml:math id="mml-ieqn-92"><mml:msub><mml:mi>Y</mml:mi><mml:mrow><mml:mtext>out</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> mitigates small-object scarcity. Positioning CSPStage before the detection head aligns task features and bridges semantic gaps. Under dense occlusions, the shallow branch bypasses deep non-linearities to preserve high-resolution textures of unoccluded fragments, while the deep branch provides robust semantic context. Aggregating these streams enables MFCI-YOLO to infer partially visible objects from surviving local textures and global priors, significantly reducing missed detections.</p>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Wise-IoU Loss Function</title>
<p>To improve localization robustness against low-quality annotations, we employ Wise-IoU v3. It first constructs a geometric distance penalty <inline-formula id="ieqn-93"><mml:math id="mml-ieqn-93"><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>W</mml:mi><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>exp</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msup><mml:mi>&#x03C1;</mml:mi><mml:mn>2</mml:mn></mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>W</mml:mi><mml:mi>g</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>H</mml:mi><mml:mi>g</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x2217;</mml:mo></mml:msup></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, where <inline-formula id="ieqn-94"><mml:math id="mml-ieqn-94"><mml:mi>&#x03C1;</mml:mi></mml:math></inline-formula> is the Euclidean distance between predicted and ground-truth centers, and <inline-formula id="ieqn-95"><mml:math id="mml-ieqn-95"><mml:msub><mml:mi>W</mml:mi><mml:mi>g</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:math></inline-formula> are the minimum bounding box dimensions illustrated in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>. The asterisk (&#x002A;) denotes detachment from the computation graph, ensuring it acts purely as an attention weight without hindering small-object convergence, yielding the base loss <inline-formula id="ieqn-96"><mml:math id="mml-ieqn-96"><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>W</mml:mi><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>v</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>W</mml:mi><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi></mml:mrow></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Illustration of geometric definition.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-5.tif"/>
</fig>
<p>To optimize gradient allocation across inconsistent annotation qualities, we introduce an outlier degree <inline-formula id="ieqn-97"><mml:math id="mml-ieqn-97"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>W</mml:mi><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>v</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mover><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi></mml:mrow></mml:msub><mml:mo accent="false">&#x00AF;</mml:mo></mml:mover></mml:math></inline-formula> (with <inline-formula id="ieqn-98"><mml:math id="mml-ieqn-98"><mml:mover><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi></mml:mrow></mml:msub><mml:mo accent="false">&#x00AF;</mml:mo></mml:mover></mml:math></inline-formula> as the batch&#x2019;s moving average loss) and a dynamic focusing gain <inline-formula id="ieqn-99"><mml:math id="mml-ieqn-99"><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mi>&#x03B2;</mml:mi><mml:mrow><mml:mi>&#x03B4;</mml:mi><mml:msup><mml:mi>&#x03B1;</mml:mi><mml:mrow><mml:mi>&#x03B2;</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B4;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:math></inline-formula>, producing the final loss <inline-formula id="ieqn-100"><mml:math id="mml-ieqn-100"><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>W</mml:mi><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>v</mml:mi><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>r</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>W</mml:mi><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>v</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. As depicted by the non-monotonic curve in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>, this mechanism acts as an intelligent dynamic filter. Calculus analysis indicates <inline-formula id="ieqn-101"><mml:math id="mml-ieqn-101"><mml:mi>r</mml:mi></mml:math></inline-formula> peaks at <inline-formula id="ieqn-102"><mml:math id="mml-ieqn-102"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>ln</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula>, effectively down-weighting both &#x201C;easy&#x201D; high-quality samples (<inline-formula id="ieqn-103"><mml:math id="mml-ieqn-103"><mml:mi>&#x03B2;</mml:mi><mml:mo>&#x226A;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>) to prevent overfitting, and &#x201C;hard&#x201D; extreme outliers (<inline-formula id="ieqn-104"><mml:math id="mml-ieqn-104"><mml:mi>&#x03B2;</mml:mi><mml:mo>&#x226B;</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>) to avoid harmful gradient updates from noisy labels. Consequently, the network optimally prioritizes ordinary-quality, challenging small objects.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>Illustration of dynamic mechanism dataset.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-6.tif"/>
</fig>
</sec>
<sec id="s3_6">
<label>3.6</label>
<title>Multi-Scale Convergence Network</title>
<p>Although YOLOv11n [<xref ref-type="bibr" rid="ref-27">27</xref>] native path aggregation network enhances feature flow, its P3&#x2013;P5 architecture causes scale mismatch and fusion loss in UAV aerial images: 32x downsampling loses small object information, and simple interpolation or convolution fails to distinguish objects from backgrounds, drowning small object features and amplifying background noise. To solve this, we propose Scale-Gradient and Context-Aware Feature Fusion Network (SG-PANet). As shown in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>, SG-PANet adopts a scale-down strategy: discarding P5 and extending fusion paths to shallow layers, constructing a P2-P3-P4 high-resolution fusion hierarchy. This preserves geometric textures and resolves small object information diffusion.</p>
<fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Schematic diagram of SG_PANet architecture.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-7.tif"/>
</fig>
<p>We introduce GCA as the fusion operator to address feature drowning. Unlike traditional interpolation (blending background and object features), GCA uses group decoupling to partition channels, dynamically enhancing object-related channels, thus injecting deep semantics into shallow features while preserving object details. Meanwhile, we replace standard convolutions with LPConv in the bottom-up localization enhancement path to reduce computational redundancy and texture noise in high-resolution features. LPConv&#x2019;s orthogonal alignment uses strip convolution branches to capture small-object morphology and additive fusion to preserve high-frequency details, reducing computation and resolving localization deviations. In summary, SG-PANet achieves performance improvement via physical-scale reconstruction and bidirectional path optimization.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experimental Process and Analysis</title>
<sec id="s4_1">
<label>4.1</label>
<title>Datasets</title>
<p>VisDrone2019 [<xref ref-type="bibr" rid="ref-28">28</xref>] (<xref ref-type="fig" rid="fig-8">Fig. 8</xref>) by Tianjin University&#x2019;s AISKYEYE team focuses on UAV-based computer vision tasks, comprising 10,209 static images (6471 training, 548 validation, 1610 testing) across 14 Chinese cities&#x2019; urban/rural scenes, with 10 core categories.</p>
<fig id="fig-8">
<label>Figure 8</label>
<caption>
<title>Label analysis diagram of the VisDrone2019 dataset.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-8.tif"/>
</fig>
<p>To validate MFCI-YOLO&#x2019;s robustness for aerial small object detection, we adopted DOTA v1.5 [<xref ref-type="bibr" rid="ref-29">29</xref>] (<xref ref-type="fig" rid="fig-9">Fig. 9</xref>)&#x2014;an upgraded DOTA v1.0 retaining 2806 images, with finer annotations and abundant &#x003C;10-pixel small instances.</p>
<fig id="fig-9">
<label>Figure 9</label>
<caption>
<title>Label analysis diagram of the DOTAv1.5 dataset.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-9.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig-8">Figs. 8</xref> and <xref ref-type="fig" rid="fig-9">9</xref> visually confirm that most UAV instances occupy sub-10-pixel ratios with severe class imbalance. This extreme distribution empirically justifies SG-PANet for preserving high-resolution details, and Wise-IoU v3 for dynamically balancing the gradients of small, low-quality samples.</p>
<p>The ablation and comparison experiments utilize the VisDrone2019 dataset (<xref ref-type="fig" rid="fig-8">Fig. 8</xref>) to validate the model&#x2019;s detection performance, while the model&#x2019;s generalization capability is evaluated using the DOTA v1.5 dataset (<xref ref-type="fig" rid="fig-9">Fig. 9</xref>).</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Experimental Environment and Evaluation Indicators</title>
<p>The experiments were conducted using an NVIDIA RTX 5090 GPU (32 GB VRAM) with CUDA 12.8 and PyTorch 2.8.0. The model was trained for up to 200 epochs (with early stopping upon convergence) using the SGD optimizer (momentum: 0.937, weight decay: 0.0005) and a Cosine Annealing scheduler with an initial learning rate of 0.01. Furthermore, to ensure statistical stability and reproducibility, all reported evaluation metrics represent the average results from three independent training runs under identical hyperparameter settings.</p>
<p>We comprehensively evaluate the model&#x2019;s detection performance using standard accuracy metrics: Precision (P), Recall (R), mAP50, and mAP50-95. Furthermore, to strictly assess the model&#x2019;s deployability on resource-constrained UAV edge devices, we measure computational efficiency using total parameters (M), computational complexity (GFLOPs), and real-time inference speed (FPS).</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Ablation Experiments</title>
<p>To systematically validate the effectiveness of each component within the MFCI-YOLO framework and their respective contributions to small object detection performance in UAV aerial photography, we designed a layer-by-layer ablation study on the VisDrone2019 dataset. Using YOLOv11n as the baseline model, we progressively integrated the Wise-IoU v3, SG-PANet, LPConv, GCA, and CSPStage modules. We quantitatively analyzed the trade-offs among accuracy (mAP), number of parameters, and FPS for each module. The quantitative results are shown in <xref ref-type="table" rid="table-1">Table 1</xref>.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Ablation experiment.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>YOLOv11n</th>
<th>WIoU v3</th>
<th>SG-PANet</th>
<th>LPConv</th>
<th>GCA</th>
<th>CSPstage</th>
<th>mAP50 (%)</th>
<th>mAP50-95 (%)</th>
<th>P (%)</th>
<th>R (%)</th>
<th>Params (M)</th>
<th>GFLOPS</th>
<th>FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td>&#x2713;</td>
<td></td>
<td></td>
<td></td>
<td></td>
<td></td>
<td>32.1</td>
<td>18.7</td>
<td>42.3</td>
<td>32.6</td>
<td>2.58</td>
<td>6.3</td>
<td>194</td>
</tr>
<tr>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td></td>
<td></td>
<td></td>
<td></td>
<td>32.4</td>
<td>18.8</td>
<td>42.6</td>
<td>33.4</td>
<td>2.58</td>
<td>6.3</td>
<td>194</td>
</tr>
<tr>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td></td>
<td></td>
<td></td>
<td>47.3</td>
<td>29.1</td>
<td>56.8</td>
<td>45.9</td>
<td>4.79</td>
<td>42.4</td>
<td>168</td>
</tr>
<tr>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td></td>
<td></td>
<td>47.0</td>
<td>28.9</td>
<td>56.7</td>
<td>44.2</td>
<td>3.38</td>
<td>38.5</td>
<td>136</td>
</tr>
<tr>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td></td>
<td>47.7</td>
<td>29.3</td>
<td>57.3</td>
<td>45.4</td>
<td>5.52</td>
<td>45.9</td>
<td>114</td>
</tr>
<tr>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td>&#x2713;</td>
<td><bold>48.2</bold></td>
<td><bold>29.8</bold></td>
<td><bold>55.7</bold></td>
<td><bold>45.0</bold></td>
<td><bold>4.50</bold></td>
<td><bold>45.0</bold></td>
<td>97</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-1fn1" fn-type="other">
<p>Note: Bold formatting indicates the optimal performance values.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p><xref ref-type="table" rid="table-1">Table 1</xref> demonstrates that the baseline YOLOv11n struggles with deep downsampling-induced feature diffusion, yielding only 32.1% mAP50. Integrating SG-PANet drastically boosts mAP50 to 47.3% by preserving high-resolution details. Replacing standard convolutions with LPConv effectively cuts parameters to 3.38M and GFLOPs to 38.5 with negligible accuracy drop, as its anisotropic receptive field perfectly matches elongated aerial objects. Furthermore, introducing GCA suppresses background noise, and CSPStage mitigates deep feature decay via historical state awareness. Ultimately, MFCI-YOLO achieves an optimal 48.2% mAP50 and 29.8% mAP50-95 at a real-time 97 FPS, significantly outperforming the baseline across all extremely small object categories (e.g., Pedestrian and Bicycle AP improved by 21.3% and 14.1%, respectively).</p>

<p>The training process visualized in <xref ref-type="fig" rid="fig-10">Fig. 10</xref> shows that MFCI-YOLO achieves a higher growth rate and superior final steady-state values for both mAP50 and mAP50-95 compared to YOLOv11n. Regarding loss functions, MFCI-YOLO demonstrates faster convergence and more stable final values in classification (Cls Loss) and distribution focal loss (DFL Loss). This confirms that the synergy of SG-PANet and Wise-IoU v3 effectively optimizes gradient propagation and enhances learning efficiency for small objects.</p>
<fig id="fig-10">
<label>Figure 10</label>
<caption>
<title>Visualization of the performance comparison between the basic model and the improved model.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-10.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig-11">Fig. 11</xref>&#x2019;s visualization offers intuitive proof. The baseline has many blue missed detections in dense crowds and distant vehicles, while MFCI-YOLO corrects these to green true detections. In summary, MFCI-YOLO optimally balances accuracy and robustness for UAV aerial scenarios under limited computation.</p>
<fig id="fig-11">
<label>Figure 11</label>
<caption>
<title>Comparison of the detection diagram of the basic model and the improved model. The left base model, the right improvement model. The green box is correctly detected, the blue box is not detected, and the red box is falsely detected.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-11.tif"/>
</fig>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Comparison Experiments</title>
<sec id="s4_4_1">
<label>4.4.1</label>
<title>Convolutional Comparison Experiment</title>
<p><xref ref-type="table" rid="table-2">Table 2</xref> shows LPConv has notable lightweight advantages. Compared to baseline standard convolution (Conv), LPConv reduces model parameters from 4.79 million to 3.38 million and GFLOPs from 42.4 to 38.5, with only a 0.3% mAP50 drop and nearly unchanged mAP50-95. We specifically selected pConv (Partial Convolution) as a baseline because it represents a state-of-the-art generic lightweight operator focused on spatial redundancy reduction. This comparison critically highlights that for UAV imagery, the domain-specific anisotropic alignment of our LPConv is superior to generic redundancy-reduction approaches. Compared with PConv, LPConv cuts parameters by an extra 18% while maintaining higher accuracy, validating its orthogonal center alignment mechanism.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Convolutional comparison table (WIou&#x002B;SG_PANet).</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Category</th>
<th>Conv</th>
<th>PConv</th>
<th>LPConv</th>
</tr>
</thead>
<tbody>
<tr>
<td>P</td>
<td>0.568</td>
<td>0.557</td>
<td>0.567</td>
</tr>
<tr>
<td>R</td>
<td>0.459</td>
<td>0.453</td>
<td>0.442</td>
</tr>
<tr>
<td>mAP50</td>
<td>0.473</td>
<td>0.473</td>
<td>0.470</td>
</tr>
<tr>
<td>mAP50-95</td>
<td>0.291</td>
<td>0.291</td>
<td>0.289</td>
</tr>
<tr>
<td>Params (M)</td>
<td>4.79</td>
<td>4.14</td>
<td>3.38</td>
</tr>
<tr>
<td>GFLOPS</td>
<td>42.4</td>
<td>45.3</td>
<td>38.5</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="table" rid="table-3">Table 3</xref> presents ablation results for core hyperparameter <italic>k</italic>. Accuracy rises steadily as <italic>k</italic> increases from 3 to 7 (expanded receptive field supplements contextual semantics), but declines when <italic>k</italic> reaches 9 or 11 (excessively large kernels introduce background noise, diluting features in sparse aerial objects). Thus, <italic>k</italic> &#x003D; 7 is selected as optimal, balancing lightweight design and feature representation capability.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Comparison table of convolution lengths of different bars of LPConv (WIoU).</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th><inline-formula id="ieqn-105"><mml:math id="mml-ieqn-105"><mml:mi>k</mml:mi></mml:math></inline-formula></th>
<th>mAP50</th>
<th>mAP50-95</th>
<th><inline-formula id="ieqn-106"><mml:math id="mml-ieqn-106"><mml:mi>P</mml:mi></mml:math></inline-formula></th>
<th><inline-formula id="ieqn-107"><mml:math id="mml-ieqn-107"><mml:mi>R</mml:mi></mml:math></inline-formula></th>
<th><inline-formula id="ieqn-108"><mml:math id="mml-ieqn-108"><mml:mi>P</mml:mi><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msup><mml:mn>10</mml:mn><mml:mn>6</mml:mn></mml:msup></mml:math></inline-formula></th>
<th>GFLOPs</th>
</tr>
</thead>
<tbody>
<tr>
<td>3</td>
<td>0.319</td>
<td>0.183</td>
<td>0.431</td>
<td>0.322</td>
<td>2.113</td>
<td>6.0</td>
</tr>
<tr>
<td>5</td>
<td>0.319</td>
<td>0.183</td>
<td>0.420</td>
<td>0.328</td>
<td>2.115</td>
<td>6.0</td>
</tr>
<tr>
<td><bold>7</bold></td>
<td><bold>0.322</bold></td>
<td><bold>0.184</bold></td>
<td><bold>0.424</bold></td>
<td><bold>0.328</bold></td>
<td><bold>2.116</bold></td>
<td><bold>6.0</bold></td>
</tr>
<tr>
<td>9</td>
<td>0.319</td>
<td>0.184</td>
<td>0.434</td>
<td>0.318</td>
<td>2.118</td>
<td>6.0</td>
</tr>
<tr>
<td>11</td>
<td>0.316</td>
<td>0.181</td>
<td>0.424</td>
<td>0.316</td>
<td>2.119</td>
<td>6.0</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-3fn1" fn-type="other">
<p>Note: Bold formatting indicates the optimal performance values.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s4_4_2">
<label>4.4.2</label>
<title>Comparison Experiments with Different Models</title>
<p>To evaluate MFCI-YOLO&#x2019;s competitiveness in UAV aerial scenarios, we conducted comparative analyses on VisDrone2019, comparing it with classic algorithms (RetinaNet, Faster R-CNN), general benchmarks (YOLOv5-YOLOv11 series), and aerial-optimized advanced algorithms (HPRS-YOLO, DI-YOLO); detailed results are in <xref ref-type="table" rid="table-4">Table 4</xref>.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Comparison table of models under VisDrone2019 validation set.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Model</th>
<th>mAP50%</th>
<th>mAP50-95%</th>
<th>P/%</th>
<th>Parameters/M</th>
<th>Gflops</th>
<th>FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td>RetinaNet</td>
<td>22.1</td>
<td>16.6</td>
<td>41.3</td>
<td>19.8</td>
<td>93.7</td>
<td>41.3</td>
</tr>
<tr>
<td>Faster R-CNN</td>
<td>33.5</td>
<td>19.3</td>
<td>45.7</td>
<td>41.2</td>
<td>206.6</td>
<td>24</td>
</tr>
<tr>
<td>YOLOv5s</td>
<td>37.9</td>
<td>22.7</td>
<td>49.1</td>
<td>9.11</td>
<td>23.7</td>
<td>133</td>
</tr>
<tr>
<td>YOLOv7-Tiny</td>
<td>35.4</td>
<td>18.9</td>
<td>45.9</td>
<td>6.33</td>
<td>13.3</td>
<td>121</td>
</tr>
<tr>
<td>YOLOv8n</td>
<td>31.9</td>
<td>18.4</td>
<td>43.2</td>
<td>3.00</td>
<td>8.1</td>
<td>232</td>
</tr>
<tr>
<td>YOLOv8s</td>
<td>39.1</td>
<td>23.3</td>
<td>50.2</td>
<td>11.2</td>
<td>28.8</td>
<td>118</td>
</tr>
<tr>
<td>YOLOv10n</td>
<td>31.7</td>
<td>18.4</td>
<td>43.3</td>
<td>2.71</td>
<td>8.4</td>
<td>241</td>
</tr>
<tr>
<td>YOLOv11n</td>
<td>32.1</td>
<td>18.7</td>
<td>42.3</td>
<td>2.58</td>
<td>6.3</td>
<td>250</td>
</tr>
<tr>
<td>YOLOv11s</td>
<td>37.8</td>
<td>22.5</td>
<td>48.0</td>
<td>9.4</td>
<td>21.3</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>HPRS-YOLO [<xref ref-type="bibr" rid="ref-18">18</xref>]</td>
<td>38.4</td>
<td>22.7</td>
<td>49.9</td>
<td>3.30</td>
<td>9.3</td>
<td>212</td>
</tr>
<tr>
<td>SBE-YOLOv8s [<xref ref-type="bibr" rid="ref-19">19</xref>]</td>
<td>42.1</td>
<td>24.3</td>
<td>51.7</td>
<td>6.2</td>
<td>105.4</td>
<td>28</td>
</tr>
<tr>
<td>DI-YOLO [<xref ref-type="bibr" rid="ref-20">20</xref>]</td>
<td>47.1</td>
<td>29.0</td>
<td>&#x2013;</td>
<td>26.14</td>
<td>96.3</td>
<td>113.4</td>
</tr>
<tr>
<td>CM-YOLOv8s [<xref ref-type="bibr" rid="ref-21">21</xref>]</td>
<td>45.9</td>
<td>27.8</td>
<td>55.4</td>
<td>3.49</td>
<td>31.2</td>
<td>72</td>
</tr>
<tr>
<td>MPAM-YOLO [<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
<td>43.1</td>
<td>26.9</td>
<td>45.5</td>
<td>41.5</td>
<td>46.5</td>
<td>52.4</td>
</tr>
<tr>
<td>SE-MSPA-DETR [<xref ref-type="bibr" rid="ref-23">23</xref>]</td>
<td>40.1</td>
<td>23.2</td>
<td>&#x2013;</td>
<td>16.03</td>
<td>65.4</td>
<td>78.5</td>
</tr>
<tr>
<td><bold>Ours</bold></td>
<td><bold>48.2</bold></td>
<td><bold>29.8</bold></td>
<td><bold>55.7</bold></td>
<td><bold>4.50</bold></td>
<td><bold>45.0</bold></td>
<td><bold>97</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="table-4fn1" fn-type="other">
<p>Note: Bold formatting indicates the optimal performance values.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Experimental results confirm MFCI-YOLO&#x2019;s architectural superiority, abandoning performance enhancement via pure parameter stacking. vs. baseline YOLOv11n, MFCI-YOLO boosts mAP50 to 48.2% while remaining lightweight via SG-PANet/GCA-based feature alignment. Notably, vs. higher-parameter YOLOv11s, MFCI-YOLO achieves a 10.4% accuracy gain with &#x003C;50% parameters, validating that small-object feature refinement outperforms blind network depth scaling.</p>
<p>MFCI-YOLO demonstrates superior performance over state-of-the-art algorithms. Compared to DI-YOLO, it achieves 48.2% mAP50 with only 1/6 the parameters and 1/2 the FLOPs. It significantly outperforms MPAM-YOLO despite having 9.2 times fewer parameters. Furthermore, unlike the resource-heavy Transformer-based hybrid framework SE-MSPA-DETR, MFCI-YOLO maintains a high-speed 97 FPS while achieving an 8.1% higher mAP50. Overall, MFCI-YOLO provides an optimal balance of accuracy and efficiency for resource-constrained UAV deployment.</p>
<p>To intuitively verify the effectiveness of the proposed modules in enhancing contextual information and suppressing background noise, we generated feature activation heatmaps for both the baseline model and MFCI-YOLO, as shown in <xref ref-type="fig" rid="fig-12">Fig. 12</xref>. The visualization captures a highly challenging night-time plaza scenario with dense, extremely small pedestrian objects.</p>
<fig id="fig-12">
<label>Figure 12</label>
<caption>
<title>Heatmap visualization comparison in a dense and low-light scenario.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-12.tif"/>
</fig>
<p>As detailed in the red zoom-in boxes, the baseline heatmap (left) exhibits diffuse activation. Its feature responses for distant small objects blur into the dark background, causing severe &#x201C;feature drowning&#x201D;. Conversely, the MFCI-YOLO heatmap (right) displays highly focused red spots strictly centered on the objects. The magnified view confirms that our model accurately activates individual tiny targets without falsely highlighting the surrounding pavement. This visual comparison verifies that LPConv&#x2019;s anisotropic receptive field effectively captures object-specific contexts, while the GCA operator cleanly decouples fine-grained features from complex background noise, successfully preserving the signals of extremely small objects.</p>
</sec>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Generalization Experiments</title>
<p>To validate robustness, we evaluated the model on the challenging DOTA v1.5 dataset. As shown in <xref ref-type="table" rid="table-5">Table 5</xref>, our method achieves an outstanding 80.3% mAP50.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Generalization experiments of different models on the DOTAv1.5 dataset.</title>
</caption>
<table>
<colgroup>
<col align="center"/>
<col align="center"/>
<col align="center"/>
<col align="center"/> </colgroup>
<thead>
<tr>
<th>Model</th>
<th>mAP50 (%)</th>
<th>Para (M)</th>
<th>GFLOPs</th>
</tr>
</thead>
<tbody>
<tr>
<td>YOLOv5n</td>
<td>62.5</td>
<td>1.8</td>
<td>4.2</td>
</tr>
<tr>
<td>YOLOv8n</td>
<td>65.6</td>
<td>3.0</td>
<td>8.1</td>
</tr>
<tr>
<td>YOLOv11n</td>
<td>75.3</td>
<td>2.6</td>
<td>6.3</td>
</tr>
<tr>
<td>BRSTD [<xref ref-type="bibr" rid="ref-30">30</xref>]</td>
<td>65.4</td>
<td>1.8</td>
<td>64.3</td>
</tr>
<tr>
<td>AG-YOLO [<xref ref-type="bibr" rid="ref-31">31</xref>]</td>
<td>72.8</td>
<td>16.6</td>
<td>&#x2013;</td>
</tr>
<tr>
<td>AESOD [<xref ref-type="bibr" rid="ref-32">32</xref>]</td>
<td>48.9</td>
<td>2.7</td>
<td>27.4</td>
</tr>
<tr>
<td>ours</td>
<td>80.3</td>
<td>4.5</td>
<td>45.0</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>DOTA v1.5 presents extreme challenges: arbitrary orientations, massive scale variations, and dense sub-10-pixel clusters. MFCI-YOLO directly addresses these: LPConv&#x2019;s orthogonal branches capture slender geometries, while GCA&#x2019;s independent high-frequency kernels suppress severe background interference, effectively preventing dense objects from drowning.</p>
<p><xref ref-type="fig" rid="fig-13">Fig. 13</xref> corroborates this analysis. Even under extreme lighting and occlusion, MFCI-YOLO achieves 0.894 precision for swimming pools and minimizes missed detections in dense vehicle areas. This confirms that CSPStage effectively bridges the semantic gap via historical state awareness. Overall, MFCI-YOLO demonstrates exceptional robustness and practical value for complex aerial deployment.</p>
<fig id="fig-13">
<label>Figure 13</label>
<caption>
<title>Visualization of DOTAv1.5 detection.</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_80341-fig-13.tif"/>
</fig>
<p>To highlight our network&#x2019;s advantages, <xref ref-type="table" rid="table-5">Table 5</xref> compares MFCI-YOLO against mainstream baselines and recent state-of-the-art detectors on the DOTAv1.5 dataset. Earlier models like YOLOv5n and YOLOv8n achieve only 62.5% and 65.6% mAP50. Recent advanced networks like BRSTD, AG-YOLO, and AESOD also struggle with dense, arbitrary-oriented targets, scoring 65.4%, 72.8%, and 48.9%, respectively. While the baseline YOLOv11n reaches 75.3%, it remains limited by feature drowning in complex scenes. In contrast, MFCI-YOLO achieves an exceptional 80.3% mAP50. Notably, it outperforms AG-YOLO by 7.5% using nearly one-fourth of its parameters. Although this precision involves a parameter increase (from 2.6M to 4.5M), our model strictly satisfies the 5M lightweight constraint. This trade-off confirms its superior accuracy, robust generalization, and essential lightweight profile for complex aerial imagery.</p>

<p>Although MFCI-YOLO exhibits exceptional generalization in dense scenarios, it still occasionally encounters missed detections and false positives under extreme physical conditions. Specifically, during severe motion blur caused by high-speed UAV maneuvering, the structural contours of extremely small objects become severely distorted, causing the anisotropic receptive field of LPConv to fail in feature alignment. Additionally, under heavy nighttime glare or extreme low-light conditions, background noise and object textures become virtually indistinguishable, occasionally bypassing the GCA operator&#x2019;s decoupling mechanism. Addressing these extreme multi-degradation scenarios via multi-modal fusion or semi-supervised learning remains an important avenue for our future research.</p>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusion</title>
<p>We proposed MFCI-YOLO, a lightweight UAV small-object detector, to address feature drowning and missed detections in complex backgrounds. Functionally, LPConv captures slender object morphologies with minimal computational redundancy, while the GCA-equipped SG-PANet decouples object signals from noise and preserves high-resolution geometric details. Integrated with CSPStage and Wise-IoU v3, the model successfully bridges semantic gaps and optimizes dynamic gradient allocation. Experimental results confirm its superiority: achieving 48.2% mAP50 at 97 FPS (only 4.50M parameters) on VisDrone2019, and 80.3% mAP50 on DOTAv1.5. Despite its exceptional generalization, future research will focus on integrating multi-modal data (e.g., infrared imaging) and semi-supervised learning to mitigate occasional missed detections under extreme motion blur or severe low-light conditions.</p>
</sec>
</body>
<back>
<ack>
<p>Not applicable.</p>
</ack>
<sec>
<title>Funding Statement: </title>
<p>This work was supported in part by the Natural Science Foundation of Henan Province under Grant 252300423317, the Science and Technology Research Project of Henan Province under Grant 262102211081, the Key Scientific Research Projects of Colleges and Universities in Henan Province under Grant 25B510012 and &#x201C;Pioneer&#x201D; and &#x201C;Leading Goose&#x201D; R&#x0026;D Program of Zhejiang under grant 2026LDC01003(JT).</p>
</sec>
<sec>
<title>Author Contributions: </title>
<p>The authors confirm their respective contributions to the paper as follows: Weiguang Wang and Jincai Li: Conceptualization, Methodology, Software, Writing&#x2014;Original Draft. Mengqi Liu: Validation, Data Curation, Investigation. Mengke Liu and Yuan Zhang: Formal Analysis, Visualization. Jingyan Wu and Yang Liu: Supervision, Project Administration, Funding Acquisition, Writing&#x2014;Review &#x0026; Editing. Junbin Lou and Yixin He: Resources, Validation. All authors reviewed and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability">
<title>Availability of Data and Materials: </title>
<p>The VisDrone2019 and DOTAv1.5 datasets analyzed in this study are publicly available. The code and models that support the findings of this study are available from the corresponding authors upon reasonable request.</p>
</sec>
<sec>
<title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement">
<title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>He</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>F</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Emergency communications in post-disaster scenarios: IoT-enhanced airship and buffer support</article-title>. <source>IEEE Internet Things J</source>. <year>2025</year>;<volume>12</volume>(<issue>9</issue>):<fpage>11457</fpage>&#x2013;<lpage>68</lpage>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>He</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>F</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Uplink outage probability analysis of AAV and intelligent connected vehicle cooperative communication using full-duplex MIMO</article-title>. <source>IEEE Commun Lett</source>. <year>2025</year>;<volume>29</volume>(<issue>9</issue>):<fpage>2068</fpage>&#x2013;<lpage>72</lpage>. doi:<pub-id pub-id-type="doi">10.1109/lcomm.2025.3585337</pub-id>; <pub-id pub-id-type="pmid">25079929</pub-id></mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>He</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>F</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>R</given-names></string-name></person-group>. <article-title>Delay minimization for NOMA-MEC offloading in ABS-aided maritime communication networks</article-title>. <source>IEEE Trans Veh Technol</source>. <year>2025</year>;<volume>74</volume>(<issue>6</issue>):<fpage>9577</fpage>&#x2013;<lpage>90</lpage>. doi:<pub-id pub-id-type="doi">10.1109/tvt.2025.3539335</pub-id>; <pub-id pub-id-type="pmid">25079929</pub-id></mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>D</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>H</given-names></string-name>, <string-name><surname>He</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>L</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Enhanced ISAC framework for moving target assisted by beyond-diagonal RIS: accurate localization and efficient communication</article-title>. <source>IEEE Trans Netw Sci Eng</source>. <year>2025</year>;<volume>12</volume>(<issue>5</issue>):<fpage>4299</fpage>&#x2013;<lpage>315</lpage>. doi:<pub-id pub-id-type="doi">10.1109/tnse.2025.3571278</pub-id>; <pub-id pub-id-type="pmid">25079929</pub-id></mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Pan</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhou</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wei</surname> <given-names>H</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>SOD-YOLO: small-object-detection algorithm based on improved YOLOv8 for UAV images</article-title>. <source>Remote Sens</source>. <year>2024</year>;<volume>16</volume>(<issue>16</issue>):<fpage>3057</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs16163057</pub-id>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wan</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Lan</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Shang</surname> <given-names>K</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>F</given-names></string-name></person-group>. <article-title>DAU-YOLO: a lightweight and effective method for small object detection in UAV images</article-title>. <source>Remote Sens</source>. <year>2025</year>;<volume>17</volume>(<issue>10</issue>):<fpage>1768</fpage>. doi:<pub-id pub-id-type="doi">10.3390/rs17101768</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yu</surname> <given-names>W</given-names></string-name>, <string-name><surname>Mo</surname> <given-names>K</given-names></string-name></person-group>. <article-title>YOLO-GCOF: a lightweight low-altitude drone detection model</article-title>. <source>IEEE Access</source>. <year>2025</year>;<volume>13</volume>(<issue>10</issue>):<fpage>53053</fpage>&#x2013;<lpage>64</lpage>. doi:<pub-id pub-id-type="doi">10.1109/access.2025.3553477</pub-id>; <pub-id pub-id-type="pmid">25079929</pub-id></mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Jian</surname> <given-names>J</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>K</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Optical remote sensing ship recognition and classification based on improved YOLOv5</article-title>. <source>Remote Sens</source>. <year>2023</year>;<volume>15</volume>(<issue>17</issue>):<fpage>4319</fpage>. doi:<pub-id pub-id-type="doi">10.20944/preprints202307.0150.v1</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Ma</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>H</given-names></string-name></person-group>. <article-title>Improved multiscale feature fusion and small object detection layer optimization for UAV aerial small object detection based on YOLOv8</article-title>. In: <conf-name>Proceedings of the International Conference on Advances in Computer Vision Research and Applications (ACVRA 2025); 2025 Feb 28&#x2013;Mar 2</conf-name>; <publisher-loc>Nanjing, China</publisher-loc>. p. <fpage>213</fpage>&#x2013;<lpage>7</lpage>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lai</surname> <given-names>D</given-names></string-name>, <string-name><surname>Kang</surname> <given-names>K</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>K</given-names></string-name>, <string-name><surname>Ma</surname> <given-names>X</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>F</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>Enhancing UAV object detection with an efficient multi-scale feature fusion framework</article-title>. <source>PLoS One</source>. <year>2025</year>;<volume>20</volume>(<issue>10</issue>):<fpage>e0332408</fpage>. doi:<pub-id pub-id-type="doi">10.1371/journal.pone.0332408</pub-id>; <pub-id pub-id-type="pmid">41060991</pub-id></mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Qiao</surname> <given-names>S</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>LC</given-names></string-name>, <string-name><surname>Yuille</surname> <given-names>A</given-names></string-name></person-group>. <article-title>DetectoRS: detecting objects with recursive feature pyramid and switchable atrous convolution</article-title>. In: <conf-name>Proceedings of the 2021 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2021 Jun 20&#x2013;25</conf-name>; <publisher-loc>Nashville, TN, USA</publisher-loc>. p. <fpage>10208</fpage>&#x2013;<lpage>19</lpage>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Sun</surname> <given-names>H</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>R</given-names></string-name>, <string-name><surname>Li</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Yang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>S</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>X</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>SET: spectral enhancement for tiny object detection</article-title>. In: <conf-name>Proceedings of the 2025 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2025 Jun 10&#x2013;17</conf-name>; <publisher-loc>Nashville, TN, USA</publisher-loc>. p. <fpage>4713</fpage>&#x2013;<lpage>23</lpage>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ye</surname> <given-names>D</given-names></string-name>, <string-name><surname>Li</surname> <given-names>G</given-names></string-name></person-group>. <article-title>A Small object detection model based on improved YOLO</article-title>. <source>J Agric Big Data</source>. <year>2025</year>;<volume>7</volume>(<issue>2</issue>):<fpage>173</fpage>&#x2013;<lpage>82</lpage>. (In Chinese). doi:<pub-id pub-id-type="doi">10.19788/j.issn.2096-6369.000073</pub-id>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ju</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Shui</surname> <given-names>J</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>J</given-names></string-name></person-group>. <article-title>GLDS-YOLO: an improved lightweight model for small object detection in UAV aerial imagery</article-title>. <source>Electronics</source>. <year>2025</year>;<volume>14</volume>(<issue>19</issue>):<fpage>3831</fpage>. doi:<pub-id pub-id-type="doi">10.3390/electronics14193831</pub-id>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhao</surname> <given-names>B</given-names></string-name>, <string-name><surname>Zhao</surname> <given-names>J</given-names></string-name>, <string-name><surname>Song</surname> <given-names>R</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Enhanced YOLO11 for lightweight and accurate drone-based maritime search and rescue object detection</article-title>. <source>PLoS One</source>. <year>2025</year>;<volume>20</volume>(<issue>7</issue>):<fpage>e0321920</fpage>. doi:<pub-id pub-id-type="doi">10.1371/journal.pone.0321920</pub-id>; <pub-id pub-id-type="pmid">40591585</pub-id></mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ji</surname> <given-names>CL</given-names></string-name>, <string-name><surname>Yu</surname> <given-names>T</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>P</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>F</given-names></string-name>, <string-name><surname>Yuan</surname> <given-names>RY</given-names></string-name></person-group>. <article-title>Yolo-tla: an efficient and lightweight small object detection model based on YOLOv5</article-title>. <source>J Real Time Image Process</source>. <year>2024</year>;<volume>21</volume>(<issue>4</issue>):<fpage>141</fpage>. doi:<pub-id pub-id-type="doi">10.1007/s11554-024-01519-4</pub-id>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhu</surname> <given-names>L</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Ke</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Lau</surname> <given-names>R</given-names></string-name></person-group>. <article-title>BiFormer: vision transformer with bi-level routing attention</article-title>. In: <conf-name>Proceedings of the 2023 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR); 2023 Jun 17&#x2013;24</conf-name>; <publisher-loc>Vancouver, BC, Canada</publisher-loc>. p. <fpage>10323</fpage>&#x2013;<lpage>33</lpage>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>W</given-names></string-name>, <string-name><surname>Gao</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>Real-time target detection algorithm for low altitude UAVs</article-title>. <source>Acta Aeronautica et Astronautica Sinica</source>. <year>2025</year>;<volume>46</volume>(<issue>16</issue>):<fpage>210</fpage>&#x2013;<lpage>23</lpage>. (In Chinese). doi:<pub-id pub-id-type="doi">10.7527/S1000-6893.2025.31619</pub-id>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Feng</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Guo</surname> <given-names>X</given-names></string-name>, <string-name><surname>Yan</surname> <given-names>J</given-names></string-name></person-group>. <article-title>Small UVA target detection algorithm based on multi-scale attention mechanism</article-title>. <source>Acta Armamentarii</source>. <year>2025</year>;<volume>46</volume>(<issue>1</issue>):<fpage>12</fpage>&#x2013;<lpage>21</lpage>. (In Chinese). doi:<pub-id pub-id-type="doi">10.12382/bgxb.2023.1124</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ding</surname> <given-names>H</given-names></string-name>, <string-name><surname>He</surname> <given-names>W</given-names></string-name>, <string-name><surname>Wan</surname> <given-names>J</given-names></string-name>, <string-name><surname>Shen</surname> <given-names>YH</given-names></string-name>, <string-name><surname>Cui</surname> <given-names>XH</given-names></string-name></person-group>. <article-title>DI-YOLO: an efficient small object detection framework for UAV aerial imagery</article-title>. <source>Control Decision</source>. <year>2025</year>;<volume>40</volume>(<issue>10</issue>):<fpage>3106</fpage>&#x2013;<lpage>16</lpage>. (In Chinese). doi:<pub-id pub-id-type="doi">10.13195/j.kzyjc.2025.0425</pub-id>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Liao</surname> <given-names>NS</given-names></string-name>, <string-name><surname>Cao</surname> <given-names>TX</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>KY</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>M</given-names></string-name>, <string-name><surname>Gu</surname> <given-names>YX</given-names></string-name>, <etal>et al.</etal></person-group> <article-title>Small target detection algorithm for UAV based on composite feature and multi-scale fusion</article-title>. <source>Comput Eng Appl</source>. <year>2025</year>;<volume>61</volume>(<issue>3</issue>):<fpage>111</fpage>&#x2013;<lpage>20</lpage>. (In Chinese). doi:<pub-id pub-id-type="doi">10.3778/j.issn.1002-8331.2407-0520</pub-id>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Fan</surname> <given-names>W</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>A multi-stage path aggregation module for small object detection on drone-captured scenarios</article-title>. <source>Digit Signal Process</source>. <year>2026</year>;<volume>173</volume>:<fpage>105901</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.dsp.2026.105901</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Li</surname> <given-names>H</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>H</given-names></string-name></person-group>. <article-title>A spatially enhanced multiscale polarity sensing framework for UAV small target detection</article-title>. <source>Appl Soft Comput</source>. <year>2026</year>;<volume>186</volume>(<issue>20</issue>):<fpage>114248</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.asoc.2025.114248</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>K</given-names></string-name>, <string-name><surname>Xu</surname> <given-names>R</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Loy</surname> <given-names>CC</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>D</given-names></string-name></person-group>. <article-title>CARAFE: content-aware reassembly of features</article-title>. In: <conf-name>Proceedings of the 2019 IEEE/CVF International Conference on Computer Vision (ICCV); 2019 Oct 27&#x2013;Nov 2</conf-name>; <publisher-loc>Seoul, Republic of Korea</publisher-loc>. p. <fpage>3007</fpage>&#x2013;<lpage>16</lpage>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Jiang</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Tan</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Wang</surname> <given-names>J</given-names></string-name>, <string-name><surname>Sun</surname> <given-names>X</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>M</given-names></string-name>, <string-name><surname>Li</surname> <given-names>H</given-names></string-name></person-group>. <article-title>GiraffeDet: a heavy-neck paradigm for object detection</article-title>. <comment>arXiv:2202.04256. 2022</comment>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>CY</given-names></string-name>, <string-name><surname>Mark Liao</surname> <given-names>HY</given-names></string-name>, <string-name><surname>Wu</surname> <given-names>YH</given-names></string-name>, <string-name><surname>Chen</surname> <given-names>PY</given-names></string-name>, <string-name><surname>Hsieh</surname> <given-names>JW</given-names></string-name>, <string-name><surname>Yeh</surname> <given-names>IH</given-names></string-name></person-group>. <article-title>CSPNet: a new backbone that can enhance learning capability of CNN</article-title>. In: <conf-name>Proceedings of the 2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW); 2020 Jun 14&#x2013;19</conf-name>; <publisher-loc>Seattle, WA, USA</publisher-loc>. p. <fpage>1571</fpage>&#x2013;<lpage>80</lpage>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Zhang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Ding</surname> <given-names>Q</given-names></string-name>, <string-name><surname>Dou</surname> <given-names>Y</given-names></string-name>, <string-name><surname>Ren</surname> <given-names>S</given-names></string-name></person-group>. <article-title>Accurately detecting dense small objects from aerial images by fusing local enhanced features and balanced category loss</article-title>. In: <conf-name>Proceedings of the 2024 36th Chinese Control and Decision Conference (CCDC); 2024 May 25&#x2013;27</conf-name>; <publisher-loc>Xi&#x2019;an, China</publisher-loc>. p. <fpage>5548</fpage>&#x2013;<lpage>53</lpage>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="other"><person-group person-group-type="author"><string-name><surname>Zhu</surname> <given-names>P</given-names></string-name>, <string-name><surname>Wen</surname> <given-names>L</given-names></string-name>, <string-name><surname>Bian</surname> <given-names>X</given-names></string-name>, <string-name><surname>Ling</surname> <given-names>H</given-names></string-name>, <string-name><surname>Hu</surname> <given-names>Q</given-names></string-name></person-group>. <article-title>Vision meets drones: a challenge</article-title>. <comment>arXiv:1804.07437. 2018</comment>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><surname>Xia</surname> <given-names>GS</given-names></string-name>, <string-name><surname>Bai</surname> <given-names>X</given-names></string-name>, <string-name><surname>Ding</surname> <given-names>J</given-names></string-name>, <string-name><surname>Zhu</surname> <given-names>Z</given-names></string-name>, <string-name><surname>Belongie</surname> <given-names>S</given-names></string-name>, <string-name><surname>Luo</surname> <given-names>J,</given-names></string-name> <etal>et al.</etal></person-group> <article-title>DOTA: a large-scale dataset for object detection in aerial images</article-title>. In: <conf-name>Proceedings of the 2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition; 2018 Jun 18&#x2013;23</conf-name>; <publisher-loc>Salt Lake City, UT, USA</publisher-loc>. p. <fpage>3974</fpage>&#x2013;<lpage>83</lpage>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Huang</surname> <given-names>S</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>C</given-names></string-name>, <string-name><surname>Jiang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Qu</surname> <given-names>Z</given-names></string-name></person-group>. <article-title>BRSTD: bio-inspired remote sensing tiny object detection</article-title>. <source>IEEE Trans Geosci Remote Sens</source>. <year>2024</year>;<volume>62</volume>:<fpage>1</fpage>&#x2013;<lpage>15</lpage>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wang</surname> <given-names>X</given-names></string-name>, <string-name><surname>Han</surname> <given-names>C</given-names></string-name>, <string-name><surname>Huang</surname> <given-names>L</given-names></string-name>, <string-name><surname>Nie</surname> <given-names>T</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>X</given-names></string-name>, <string-name><surname>Liu</surname> <given-names>H</given-names></string-name>, <etal>et al</etal></person-group>. <article-title>AG-yolo: attention-guided yolo for efficient remote sensing oriented object detection</article-title>. <source>Remote Sens</source>. <year>2025</year>;<volume>17</volume>(<issue>6</issue>):<fpage>1027</fpage>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Peng</surname> <given-names>J</given-names></string-name>, <string-name><surname>Lv</surname> <given-names>K</given-names></string-name>, <string-name><surname>Lin</surname> <given-names>D</given-names></string-name>, <string-name><surname>Yuan</surname> <given-names>L</given-names></string-name></person-group>. <article-title>AESOD: towards accurate and efficient general-purpose small object detection</article-title>. <source>Digit Signal Process</source>. <year>2026</year>;<volume>176</volume>(<issue>4</issue>):<fpage>106037</fpage>. doi:<pub-id pub-id-type="doi">10.1016/j.dsp.2026.106037</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>