@article{Dong2024, 
author = {Liwei Dong and Ni Li and Guanghong Gong and Xin Lin},
title = {Offline Reinforcement Learning with Constrained Hybrid Action Implicit Representation Towards Wargaming Decision-Making},
year = {2024},
journal = {Tsinghua Science and Technology},
volume = {29},
number = {5},
pages = {1422-1440},
keywords = {offline Reinforcement Learning (RL), wargaming, decision-making, hybrid action space},
url = {https://www.sciopen.com/article/10.26599/TST.2023.9010100},
doi = {10.26599/TST.2023.9010100},
abstract = {Reinforcement Learning (RL) has emerged as a promising data-driven solution for wargaming decision-making. However, two domain challenges still exist: (1) dealing with discrete-continuous hybrid wargaming control and (2) accelerating RL deployment with rich offline data. Existing RL methods fail to handle these two issues simultaneously, thereby we propose a novel offline RL method targeting hybrid action space. A new constrained action representation technique is developed to build a bidirectional mapping between the original hybrid action space and a latent space in a semantically consistent way. This allows learning a continuous latent policy with offline RL with better exploration feasibility and scalability and reconstructing it back to a needed hybrid policy. Critically, a novel offline RL optimization objective with adaptively adjusted constraints is designed to balance the alleviation and generalization of out-of-distribution actions. Our method demonstrates superior performance and generality across different tasks, particularly in typical realistic wargaming scenarios.}
}