@misc{indiciae3fd825f6a4fc, title = {Distilling Vision-Language Models on Millions of Videos}, author = {Yue Zhao and Long Zhao and Xingyi Zhou and Jialin Wu and Chun-Te Chu and Hui Miao and Florian Schroff and Hartwig Adam and Ting Liu and Boqing Gong and Philipp Krähenbühl and Liangzhe Yuan}, year = {2024}, url = {https://arxiv.org/abs/2401.06129}, note = {Source identifier: 2401.06129} }