@misc{indiciae1a5fa4d04958, title = {Analyzing Zero-Shot Abilities of Vision-Language Models on Video Understanding Tasks}, author = {Avinash Madasu and Anahita Bhiwandiwalla and Vasudev Lal}, year = {2023}, url = {https://arxiv.org/abs/2310.04914}, note = {Source identifier: 2310.04914} }