@misc{indiciae3c23f16073b2, title = {Learning Video Representations from Textual Web Supervision}, author = {Jonathan C. Stroud and Zhichao Lu and Chen Sun and Jia Deng and Rahul Sukthankar and Cordelia Schmid and David A. Ross}, year = {2021}, url = {https://arxiv.org/abs/2007.14937}, note = {Source identifier: 2007.14937} }