@misc{indiciae7918f68049cd, title = {Pretrained Image-Text Models are Secretly Video Captioners}, author = {Chunhui Zhang and Yiren Jian and Zhongyu Ouyang and Soroush Vosoughi}, year = {2025}, url = {https://arxiv.org/abs/2502.13363}, note = {Source identifier: 2502.13363} }