@misc{indiciae280b04aee9af, title = {Unifying Vision-and-Language Tasks via Text Generation}, author = {Jaemin Cho and Jie Lei and Hao Tan and Mohit Bansal}, year = {2021}, url = {https://arxiv.org/abs/2102.02779}, note = {Source identifier: 2102.02779} }