@misc{indiciae43dff7082037, title = {LXMERT: Learning Cross-Modality Encoder Representations from Transformers}, author = {Hao Tan and Mohit Bansal}, year = {2019}, url = {https://arxiv.org/abs/1908.07490}, note = {Source identifier: 1908.07490} }