@misc{indiciaedc635ea44c4b, title = {Multimodal Open-Vocabulary Video Classification via Pre-Trained Vision and Language Models}, author = {Rui Qian and Yeqing Li and Zheng Xu and Ming-Hsuan Yang and Serge Belongie and Yin Cui}, year = {2022}, url = {https://arxiv.org/abs/2207.07646}, note = {Source identifier: 2207.07646} }