@misc{indiciae4bf17ce6faca, title = {CLIP-ViP: Adapting Pre-trained Image-Text Model to Video-Language Representation Alignment}, author = {Hongwei Xue and Yuchong Sun and Bei Liu and Jianlong Fu and Ruihua Song and Houqiang Li and Jiebo Luo}, year = {2023}, url = {https://arxiv.org/abs/2209.06430}, note = {Source identifier: 2209.06430} }