@misc{indiciaef4cf4f563741, title = {Masked Vision and Language Modeling for Multi-modal Representation Learning}, author = {Gukyeong Kwon and Zhaowei Cai and Avinash Ravichandran and Erhan Bas and Rahul Bhotika and Stefano Soatto}, year = {2023}, url = {https://arxiv.org/abs/2208.02131}, note = {Source identifier: 2208.02131} }