@misc{indiciaee84e2e3fc769, title = {ViLT: Vision-and-Language Transformer Without Convolution or Region Supervision}, author = {Wonjae Kim and Bokyung Son and Ildoo Kim}, year = {2021}, url = {https://arxiv.org/abs/2102.03334}, note = {Source identifier: 2102.03334} }