@misc{indiciaea6369d7ea52c, title = {Visual Representation Matters: Exploiting Temporal Differences in Video-to-Audio Generation}, author = {Zehua Chen and Junyou Wang and Yuxuan Jiang and Zhenying Fang and Yusheng Dai and Jianfei Chen and Ziwei Liu and Jun Zhu}, year = {2026}, url = {https://arxiv.org/abs/2608.04902}, note = {Source identifier: 2608.04902} }