
@Article{cmc.2026.083306,
AUTHOR = {Xiangyi Le, Deyu Lin, Yufei Zhao, Wang Miao, Yong Liang Guan},
TITLE = {Multi-UAV Collaborative Energy Charging for Battery-Free SWIPT-Enabled Sensor Networks Based on MADDPG},
JOURNAL = {Computers, Materials \& Continua},
VOLUME = {},
YEAR = {},
NUMBER = {},
PAGES = {{pages}},
URL = {http://www.techscience.com/cmc/online/detail/27461},
ISSN = {1546-2226},
ABSTRACT = {The emergence of Unmanned Aerial Vehicle (UAV)-enabled Wireless Energy Transfer (WET) and Simultaneous Wireless Information and Power Transfer (SWIPT) technology provide a promising solution to overcome the energy sustainability limitations of traditional harvesting-reliant sensor networks. However, in large-scale Battery-free SWIPT-enabled Sensor Networks (BSSN) characterized by sparse node distribution and heterogeneous energy consumption and harvesting rates, employing a single UAV for energy replenishment often suffers from insufficient operation continuity and low charging efficiency. To overcome these challenges, a Multi-UAV Collaborative Energy Charging for BSSN Based on Multi-Agent Deep Deterministic Policy Gradient (MCEC-MADDPG) is proposed in this paper. Specifically, we construct a collaborative one-to-one precision energy supply model where UAVs hover directly above specific nodes to achieve power transmission without complex beamforming requirements. To achieve collaborative scheduling among multiple UAVs in wide-area dynamic environments, the energy replenishment problem is first formulated as a Partially Observable Markov Decision Process (POMDP). Subsequently, the Centralized Training with Decentralized Execution (CTDE) architecture of the MADDPG algorithm is leveraged to solve this POMDP, which effectively tackles the non-stationarity challenge inherent in multi-agent environments. Simulation results demonstrate that MCEC-MADDPG exhibits superior performance in terms of convergence speed and stability. It enables the adaptive emergence of spatial-division collaborative strategies, significantly enhances the average residual energy of the network, and elevates the node survival rate to nearly <mml:math id="mml-ieqn-1"><mml:mn>90</mml:mn><mml:mrow><mml:mtext>% </mml:mtext></mml:mrow></mml:math>. Compared with Deep Deterministic Policy Gradient (DDPG), the traditional static Partition-Greedy method, the heuristic K-Means algorithm and the dynamic Two-Layer task allocation strategy, the proposed approach demonstrates substantial advantages.},
DOI = {10.32604/cmc.2026.083306}
}



