@misc{indiciaed97b8d8ace21, title = {GeoVista: Visually Grounded Active Perception for Vision-Language Understanding of Ultra-High-Resolution Remote Sensing Images}, author = {Jiashun Zhu and Ronghao Fu and Jiasen Hu and Jing Huang and Nachuan Xing and Bo Yang}, year = {2026}, url = {https://arxiv.org/abs/2605.14475}, note = {Source identifier: 2605.14475} }