@misc{indiciae7ad573bdd2d2, title = {Encapsulated Composition of Text-to-Image and Text-to-Video Models for High-Quality Video Synthesis}, author = {Tongtong Su and Chengyu Wang and Bingyan Liu and Jun Huang and Dongming Lu}, year = {2025}, url = {https://arxiv.org/abs/2507.13753}, note = {Source identifier: 2507.13753} }