@misc{indiciae71f47ce8e2d1, title = {Implicit Geometry Representations for Vision-and-Language Navigation from Web Videos}, author = {Mingfei Han and Haihong Hao and Liang Ma and Kamila Zhumakhanova and Ekaterina Radionova and Jingyi Zhang and Xiaojun Chang and Xiaodan Liang and Ivan Laptev}, year = {2026}, url = {https://arxiv.org/abs/2603.09259}, note = {Source identifier: 2603.09259} }