@misc{indiciae0bc74d12b552, title = {Transfer Learning from Audio-Visual Grounding to Speech Recognition}, author = {Wei-Ning Hsu and David Harwath and James Glass}, year = {2019}, url = {https://arxiv.org/abs/1907.04355}, note = {Source identifier: 1907.04355} }