@misc{indiciaeaaf6d0d74ce3, title = {Listen, Look and Deliberate: Visual context-aware speech recognition using pre-trained text-video representations}, author = {Shahram Ghorbani and Yashesh Gaur and Yu Shi and Jinyu Li}, year = {2020}, url = {https://arxiv.org/abs/2011.04084}, note = {Source identifier: 2011.04084} }