@misc{indiciae9822cf6c4668, title = {NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation}, author = {Jiazhao Zhang and Kunyu Wang and Rongtao Xu and Gengze Zhou and Yicong Hong and Xiaomeng Fang and Qi Wu and Zhizheng Zhang and He Wang}, year = {2024}, url = {https://arxiv.org/abs/2402.15852}, note = {Source identifier: 2402.15852} }