@misc{indiciaef91781d6af75, title = {VidLA: Video-Language Alignment at Scale}, author = {Mamshad Nayeem Rizve and Fan Fei and Jayakrishnan Unnikrishnan and Son Tran and Benjamin Z. Yao and Belinda Zeng and Mubarak Shah and Trishul Chilimbi}, year = {2024}, url = {https://arxiv.org/abs/2403.14870}, note = {Source identifier: 2403.14870} }