@misc{indiciae2f418d738423, title = {Beyond Appearance: Can Multimodal Large Language Models Exploit Vertical Structure for Remote Sensing Natural Scene Understanding?}, author = {Jing Huang and Duanchu Wang and Junjie Yang and Zihang Cheng and Cheng Li and Lin Cui and Zhouyi Wu and Di Wang}, year = {2026}, url = {https://arxiv.org/abs/2605.25784}, note = {Source identifier: 2605.25784} }