@article{Ke2026, 
author = {Yan Ke and Fatemeh Shiri and Hu Zhang and Xin Yu},
title = {Thinking outside the frame: viewpoint-aware spatial reasoning in large multimodal models},
year = {2026},
journal = {Visual Intelligence},
volume = {4},
pages = {21},
keywords = {Large multimodal model (LMM), Spatial reasoning, Viewpoint, Prompt, Visual-grounding},
url = {https://www.sciopen.com/article/10.1007/s44267-026-00126-0},
doi = {10.1007/s44267-026-00126-0},
abstract = {Large multimodal models (LMMs) have greatly improved cross-modal tasks but continue to struggle with spatial reasoning. In particular, they exhibit a bias toward egocentric reasoning and have limited ability to reframe spatial relations allocentrically, making them unreliable under allocentric viewpoints. In response, we first propose a lightweight viewpoint-aware self-correction strategy with low token overhead that serves as an inductive signal, sensitizing the model to frame-of-reference changes and guiding a corresponding shift in its reasoning trajectory. Second, beyond final-answer accuracy, we introduce a multi-hop spatial reasoning trace that decomposes reasoning into a front-hop object identification stage and a back-hop spatial relation inference stage to expose intermediate reasoning states and assess whether genuine viewpoint transformations occur. Third, we further design robustness-oriented metrics, including hop degradation slope (HDS) and minimality and redundancy of multi-hop reasoning (MRR), to enable a finer-grained analysis of multi-hop reasoning behavior. Finally, we conduct comprehensive quantitative and qualitative evaluations on Spatial-CoT and COMFORT to validate the effectiveness of our approach and provide insights into robustness and failure modes.}
}