@misc{indiciae760cdf6932bc, title = {Embodied Scene Understanding for Vision Language Models via MetaVQA}, author = {Weizhen Wang and Chenda Duan and Zhenghao Peng and Yuxin Liu and Bolei Zhou}, year = {2025}, url = {https://arxiv.org/abs/2501.09167}, note = {Source identifier: 2501.09167} }