@misc{indiciae9ca8c3019859, title = {GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models}, author = {Zhangyang Qi and Zhixiong Zhang and Ye Fang and Jiaqi Wang and Hengshuang Zhao}, year = {2025}, url = {https://arxiv.org/abs/2501.01428}, note = {Source identifier: 2501.01428} }