@misc{indiciaeb6e2e0161875, title = {Augment the Pairs: Semantics-Preserving Image-Caption Pair Augmentation for Grounding-Based Vision and Language Models}, author = {Jingru Yi and Burak Uzkent and Oana Ignat and Zili Li and Amanmeet Garg and Xiang Yu and Linda Liu}, year = {2023}, url = {https://arxiv.org/abs/2311.02536}, note = {Source identifier: 2311.02536} }