@misc{indiciae6819c2d2e065, title = {VITR: Augmenting Vision Transformers with Relation-Focused Learning for Cross-Modal Information Retrieval}, author = {Yan Gong and Georgina Cosma and Axel Finke}, year = {2023}, url = {https://arxiv.org/abs/2302.06350}, note = {Source identifier: 2302.06350} }