@misc{indiciae1f9c4db89bf1, title = {AVLnet: Learning Audio-Visual Language Representations from Instructional Videos}, author = {Andrew Rouditchenko and Angie Boggust and David Harwath and Brian Chen and Dhiraj Joshi and Samuel Thomas and Kartik Audhkhasi and Hilde Kuehne and Rameswar Panda and Rogerio Feris and Brian Kingsbury and Michael Picheny and Antonio Torralba and James Glass}, year = {2021}, url = {https://arxiv.org/abs/2006.09199}, note = {Source identifier: 2006.09199} }