@misc{indiciae9bfbb4486730, title = {WIT: Wikipedia-based Image Text Dataset for Multimodal Multilingual Machine Learning}, author = {Krishna Srinivasan and Karthik Raman and Jiecao Chen and Michael Bendersky and Marc Najork}, year = {2021}, doi = {10.1145/3404835.3463257}, url = {https://arxiv.org/abs/2103.01913}, note = {Source identifier: 2103.01913} }