@article{LI2026, 
author = {Xuedong LI and Yunfeng DONG},
title = {An expert-guided hierarchical reinforcement learning method for collaborative mission planning in LEO satellite cluster},
year = {2026},
journal = {Chinese Journal of Aeronautics},
volume = {39},
number = {7},
keywords = {Expert knowledge, Satellite cluster, Uncertain systems, Mission planning, Reinforcement learning, Simulated annealing},
url = {https://www.sciopen.com/article/10.1016/j.cja.2025.103868},
doi = {10.1016/j.cja.2025.103868},
abstract = {With the increasing complexity of earth observation missions, mission planning for Low Earth Orbit (LEO) satellite cluster faces the growing contradiction between rising observation demands and limited onboard resources, which in turn intensifies environmental uncertainties and satellite system uncertainties during the observation process. To address the challenge of inefficient exploration by traditional Reinforcement Learning (RL) approaches for satellite mission planning, this paper proposes an Expert-Guided Hierarchical Reinforcement Learning (EG-HRL) method. The proposed method begins by establishing a physically-informed mission planning model that accounts for orbital perturbations from Earth’s non-spherical gravitational potential and eccentricity effects. A physics-based observation window prediction model is developed to enhance environment representation and support RL-based decision making. An integer programming model is also formulated to represent observation sequencing under task constraints and dynamic onboard resource limitations. A dual-layer EG-HRL framework is then designed. The high-level policy selects observation windows based on target priority and an expert-weighted reward function, while the low-level policy performs real-time window adjustment using a cost function derived from satellite resource states and expert-informed interval modeling. Simulation results demonstrate that the integration of expert knowledge significantly enhances planning performance in complex mission scenarios. Ablation studies confirm that both the expert guidance and the hierarchical structure are critical to this improvement. Comparative experiments further indicate that EG-HRL consistently outperforms both flat reinforcement learning and metaheuristic methods such as Improved Simulated Annealing (ISA) across various task scales. Moreover, evaluations under diverse internal and external uncertainties validate the method’s strong adaptability and robustness.}
}