@misc{indiciaee535aa878c5d, title = {TVLT: Textless Vision-Language Transformer}, author = {Zineng Tang and Jaemin Cho and Yixin Nie and Mohit Bansal}, year = {2022}, url = {https://arxiv.org/abs/2209.14156}, note = {Source identifier: 2209.14156} }