@misc{indiciae6b2971032e69, title = {CoVLR: Coordinating Cross-Modal Consistency and Intra-Modal Structure for Vision-Language Retrieval}, author = {Yang Yang and Zhongtian Fu and Xiangyu Wu and Wenjie Li}, year = {2023}, url = {https://arxiv.org/abs/2304.07567}, note = {Source identifier: 2304.07567} }