
@Article{cmc.2026.086756,
AUTHOR = {Yu Tong, Kaina Xiong, Jun Liu, Xinyue Fan, Guixing Cao},
TITLE = {FLD-RTDETR: A Frequency-Aware Lightweight Network for Small-Object Detection in Aerial Imagery},
JOURNAL = {Computers, Materials \& Continua},
VOLUME = {},
YEAR = {},
NUMBER = {},
PAGES = {{pages}},
URL = {http://www.techscience.com/cmc/online/detail/28242},
ISSN = {1546-2226},
ABSTRACT = {Small-object detection in aerial remote-sensing imagery is constrained by the cascaded low-pass behaviour induced by consecutive spatial down-sampling, which progressively attenuates the high-frequency cues of weak targets. Recent detection transformers, whose self-attention operators behave—under commonly used temperature and rank regimes—as adaptive spatial low-pass filters, tend to inherit rather than mitigate this spectral bias, while existing remedies are typically obtained at the cost of considerable parameter and floating-point overhead. To bridge this gap, this paper proposes a Frequency-Aware Lightweight Network for Small-Object Detection (FLD-RTDETR), built upon the Real-Time Detection Transformer (RT-DETR), which augments this baseline along three orthogonal axes. A Partial Channel Bottleneck (PCB) routes a fraction of channels through identity bypass to preserve un-attenuated high-frequency channels while reducing redundant floating-point operations. A Local Spatial and Global Channel Attention (LSGCA) module factorizes self-attention into orthogonal spatial-covariance and channel-covariance branches, lowering complexity to local linearity. A Lightweight Multi-scale Fusion Neck (LMFN), centred on an Omni-Dimensional Frequency–Spatial Attention (ODFA) hub, injects explicit frequency-domain compensation at the multi-scale fusion junction. On the VisDrone2019 benchmark, FLD-RTDETR attains a mean Average Precision (mAP) of 50.2% (<math id="mml-ieqn-1"><msub><mrow><mi mathvariant="normal">m</mi><mi mathvariant="normal">A</mi><mi mathvariant="normal">P</mi></mrow><mrow><mn>50</mn></mrow></msub></math>) and 30.9% (<math id="mml-ieqn-2"><msub><mrow><mi mathvariant="normal">m</mi><mi mathvariant="normal">A</mi><mi mathvariant="normal">P</mi></mrow><mrow><mn>50</mn><mo>:</mo><mn>95</mn></mrow></msub></math>) at only 20.11M parameters and 86.4 giga floating-point operations (GFLOPs), improving over the RT-DETR-R18 baseline by 3.1 and 2.1 absolute points, respectively and reaching parity with RT-DETR-R50 at less than half its parameters and roughly two-thirds of its floating-point operations. Evaluation on the structurally different NWPU VHR-10 benchmark—on which the model is separately trained and tested—further indicates robustness across scale distributions and viewpoint geometries, providing a competitive accuracy/efficiency trade-off for high-fidelity small-object detection in aerial remote sensing.},
DOI = {10.32604/cmc.2026.086756}
}



