@misc{indiciae83a0e38362ea, title = {VX2TEXT: End-to-End Learning of Video-Based Text Generation From Multimodal Inputs}, author = {Xudong Lin and Gedas Bertasius and Jue Wang and Shih-Fu Chang and Devi Parikh and Lorenzo Torresani}, year = {2021}, url = {https://arxiv.org/abs/2101.12059}, note = {Source identifier: 2101.12059} }