@misc{indiciae083c33291d56, title = {ELVIS: Empowering Locality of Vision Language Pre-training with Intra-modal Similarity}, author = {Sumin Seo and JaeWoong Shin and Jaewoo Kang and Tae Soo Kim and Thijs Kooi}, year = {2023}, url = {https://arxiv.org/abs/2304.05303}, note = {Source identifier: 2304.05303} }