@misc{indiciaeaa8a3fded6fe, title = {GMS-CAVP: Improving Audio-Video Correspondence with Multi-Scale Contrastive and Generative Pretraining}, author = {Shentong Mo and Zehua Chen and Jun Zhu}, year = {2026}, url = {https://arxiv.org/abs/2601.19606}, note = {Source identifier: 2601.19606} }