@misc{indiciaee918fee23e7b, title = {Spoken Moments: Learning Joint Audio-Visual Representations from Video Descriptions}, author = {Mathew Monfort and SouYoung Jin and Alexander Liu and David Harwath and Rogerio Feris and James Glass and Aude Oliva}, year = {2021}, url = {https://arxiv.org/abs/2105.04489}, note = {Source identifier: 2105.04489} }