@misc{indiciaee8706aaccf9c, title = {Unified Text-Image-to-Video Generation: A Training-Free Approach to Flexible Visual Conditioning}, author = {Bolin Lai and Sangmin Lee and Xu Cao and Xiang Li and James M. Rehg}, year = {2026}, url = {https://arxiv.org/abs/2505.20629}, note = {Source identifier: 2505.20629} }