@misc{indiciaee90e3906511b, title = {The Best of Both Worlds: Integrating Language Models and Diffusion Models for Video Generation}, author = {Aoxiong Yin and Kai Shen and Yichong Leng and Xu Tan and Xinyu Zhou and Juncheng Li and Siliang Tang}, year = {2025}, url = {https://arxiv.org/abs/2503.04606}, note = {Source identifier: 2503.04606} }