@misc{indiciae088a766b3e06, title = {Region-Level Context-Aware Multimodal Understanding}, author = {Hongliang Wei and Xianqi Zhang and Xingtao Wang and Xiaopeng Fan and Debin Zhao}, year = {2025}, url = {https://arxiv.org/abs/2508.12263}, note = {Source identifier: 2508.12263} }