@misc{indiciaee52978533bbd, title = {Seer: Language Instructed Video Prediction with Latent Diffusion Models}, author = {Xianfan Gu and Chuan Wen and Weirui Ye and Jiaming Song and Yang Gao}, year = {2026}, url = {https://arxiv.org/abs/2303.14897}, note = {Source identifier: 2303.14897} }