@misc{indiciaed89e8e43b246, title = {Improving Video Diffusion Transformer Training by Multi-Feature Fusion and Alignment from Self-Supervised Vision Encoders}, author = {Dohun Lee and Hyeonho Jeong and Jiwook Kim and Duygu Ceylan and Jong Chul Ye}, year = {2025}, url = {https://arxiv.org/abs/2509.09547}, note = {Source identifier: 2509.09547} }