@misc{indiciaed57931419bb8, title = {VISTA: Triplet-Supervised Video Style Transfer with Diffusion Transformers}, author = {Yiren Song and Wangzi Yao and Haofan Wang and Mike Zheng Shou}, year = {2026}, url = {https://arxiv.org/abs/2605.17312}, note = {Source identifier: 2605.17312} }