@article{YANG2026, 
author = {Bingjie YANG and Xuejun ZHANG and Zhuoya LIU and Yue XIAO},
title = {UAV Encounter Scenario Generation Method Based on Reward Shaping},
year = {2026},
journal = {Journal of South China University of Technology (Natural Science Edition)},
volume = {54},
number = {6},
pages = {183-192},
keywords = {unmanned aerial vehicle, detect-and-avoid system, reinforcement learning, reward shaping, encounter scenario generation},
url = {https://www.sciopen.com/article/10.12141/j.issn.1000-565X.250411},
doi = {10.12141/j.issn.1000-565X.250411},
abstract = {To address the encounter scenario generation problem for validating UAV Detect-and-Avoid (DAA) systems, this paper proposes an improved PPO algorithm—RSC-PPO—which is based on reward shaping and action constraints. To guarantee the physical fidelity of the scenarios, a Markov Decision Process (MDP) based on variable-speed Dubins dynamics is established. Furthermore, an action smoothing constraint regularization term is introduced to effectively suppress drastic fluctuations in policy output, ensuring that the intruder trajectories possess physical realism. Moreover, to overcome the challenges of exploration difficulty and poor convergence in sparse reward environments for reinforcement learning, a multi-level reward shaping mechanism comprising heading accuracy guidance, circling penalty, and time pressure penalty is developed. This mechanism converts the sparse terminal encounter signal into gradual and continuous process guidance signals, thereby markedly enhancing the exploration efficiency of high-risk trajectories. Experimental results demonstrate that the trained RSC-PPO policy can efficiently and stably generate highly adversarial scenario sets, achieving a success rate exceeding 95% and realizing 100% coverage of the initial state space. The generated scenario set covers multiple typical conflict configurations, including head-on, converging, and overtaking scenarios, fully ensuring the diversity of encounters. Risk assessment based on the Effective Maneuvering Time Window (EMTW) confirms that the generated scenarios exhibit significant characteristics of high urgency. This study provides an efficient and robust scenario dataset and generation framework for the validation of safety-critical aviation systems such as DAA.}
}