@misc{indiciae3224609ecee0, title = {Watching Synthetic Videos: Aligning Cross-modal Representations with Visual Synthesis for Zero-shot Video Captioning}, author = {Liangyu Fu and Junbo Wang and Yuke Li and Ya Jing and Xuecheng Wu and Zhiyong Wang}, year = {2026}, url = {https://arxiv.org/abs/2608.11013}, note = {Source identifier: 2608.11013} }