@misc{indiciaee8387ec0f92e, title = {Learning Physics from Pretrained Video Models: A Multimodal Continuous and Sequential World Interaction Models for Robotic Manipulation}, author = {Zijian Song and Qichang Li and Sihan Qin and Yuhao Chen and Tianshui Chen and Liang Lin and Guangrun Wang}, year = {2026}, url = {https://arxiv.org/abs/2603.00110}, note = {Source identifier: 2603.00110} }