@article{Yang2026, 
author = {Chao Yang and Chengbo Wei and Yiming Zhao and Liming Wang and Peigang Xu and Kunlun Qi and Yuanzheng Shao and Huayi Wu},
title = {MSALNet: a multi-scale adaptive learning network for high-resolution remote sensing scene classification},
year = {2026},
journal = {Geo-Spatial Information Science},
volume = {29},
number = {1},
pages = {143-167},
keywords = {Remote sensing scene classification, convolutional neural network (CNN), scale effect, multiscale feature fusion, scale adaptive},
url = {https://www.sciopen.com/article/10.1080/10095020.2025.2514822},
doi = {10.1080/10095020.2025.2514822},
abstract = {High-resolution remote sensing (HRS) images often feature objects of varying sizes within the same scene, presenting significant challenges for conventional CNN with fixed-size receptive fields. To address this issue, we propose a multi-scale adaptive learning network (MSALNet) that learns optimal scales in a weakly supervised manner and efficiently fuses multiscale features to enhance feature integration and representation across varying object scales. The MSALNet begins by extracting original features using dilated convolution, effectively capturing information from objects of diverse sizes. It then learns optimal scale parameters from these features to generate scale-transformed representations tailored to different scene contexts. To ensure seamless integration, the original and scale-transformed features are dynamically aligned and fused across multiple network layers. This process produces robust multiscale representations, which are subsequently processed through a fully connected layer with softmax activation for precise scene classification. Extensive experiments on the RSSCN7, AID, and NWPU-RESISC45 datasets demonstrate that MSALNet significantly outperforms traditional CNN, especially in scene categories with pronounced scale variations. These results highlight its robustness and adaptability in addressing complex HRS scenarios.}
}