@article{CHEN2025, 
author = {Lin CHEN and Xiwen GU and Zhiying CHEN and Zhuo ZHANG and Xiaoliang SUN},
title = {High-precision monocular vision pose measurement for large distance span in carrier landing guidance},
year = {2025},
journal = {Acta Aeronautica et Astronautica Sinica},
volume = {46},
number = {15},
keywords = {monocular, landing guidance, pose measurement, deep learning, keypoint detection},
url = {https://www.sciopen.com/article/10.7527/S1000-6893.2025.31568},
doi = {10.7527/S1000-6893.2025.31568},
abstract = {The autonomous landing guidance involves a large distance span, resulting in significant scale variations of the ship target in the image sequences obtained through monocular vision guidance. Existing pose measurement methods struggle to achieve high-precision monocular vision pose measurement across such a wide distance range. For current monocular vision pose measurement methods based on sparse keypoint sets, this paper focuses on improving the accuracy of keypoint detection, and analyzes the impact of target size and network input size on keypoint detection accuracy. Furthermore, this paper proposes a novel monocular vision pose measurement method based on multiple components, balancing both accuracy and efficiency. By using sparse keypoint sets to represent components in a simplified manner, and building on a coarse pose estimation of the overall ship target components, this method introduces a path aggregation feature pyramid network and a hierarchical encoding module to achieve high-precision detection of local component keypoints. Subsequently, by integrating the high-precision keypoint detection results of all components and solving the Perspective-n-Points (PnP) problem, the method achieves robust and high-precision pose measurement across the large distance span required for landing guidance. Simulation experiments and scaled physical experiments demonstrate that the proposed method achieves robust and high-precision monocular pose measurement across the large distance span for landing guidance, outperforming existing methods, with an average single-frame inference time of approximately 40 ms on embedded platforms.}
}