@article{LIU2026, 
author = {Shuyan LIU and Liu HE and Jianghui ZENG},
title = {Fine-grained semantic-enhanced cross-modal image-text retrieval method for civil aviation},
year = {2026},
journal = {Journal of Beijing University of Aeronautics and Astronautics},
volume = {52},
number = {9},
pages = {3075-3088},
keywords = {low-altitude economy, civil aviation, cross-modal retrieval, text description optimization, fine-grained semantic enhancement},
url = {https://www.sciopen.com/article/10.13700/j.bh.1001-5965.2025.0549},
doi = {10.13700/j.bh.1001-5965.2025.0549},
abstract = {In the fields of low-altitude economy and civil aviation, security assurance tasks heavily rely on the efficient correlation of cross-modal information such as images and texts. However, while mainstream cross-modal retrieval models perform well on general datasets, they underperform in these areas, which require high levels of fine-grained semantic understanding. Based on existing civil aviation datasets, a cross-modal retrieval approach with fine-grained semantic augmentation is suggested as a solution to this problem, creating a whole pipeline that includes data processing, model development, and training. First, text descriptions are optimized and enhanced based on a large multimodal model to construct a cross-modal retrieval dataset containing rich semantic information. Second, a model is created using a popular cross-modal retrieval framework. To improve the model's ability to express fine-grained semantic features, techniques such class supervision, key semantic information masking, and a fine-grained feature extraction module are introduced. Experimental results on two datasets verify the effectiveness of the proposed method, providing a reference technical path for cross-modal retrieval in low-altitude economy scenarios.}
}