@misc{indiciaef14633676907, title = {Beyond Pixels: Introducing Geometric-Semantic World Priors for Video-based Embodied Models via Spatio-temporal Alignment}, author = {Jinzhou Tang and Jusheng zhang and Sidi Liu and Waikit Xiu and Qinhan Lv and Xiying Li}, year = {2025}, url = {https://arxiv.org/abs/2509.00210}, note = {Source identifier: 2509.00210} }