@misc{indiciaeea749ce87109, title = {Training-free Guidance in Text-to-Video Generation via Multimodal Planning and Structured Noise Initialization}, author = {Jialu Li and Shoubin Yu and Han Lin and Jaemin Cho and Jaehong Yoon and Mohit Bansal}, year = {2025}, url = {https://arxiv.org/abs/2504.08641}, note = {Source identifier: 2504.08641} }