@misc{indiciae5255198fce3f, title = {Multi-modal Reference Learning for Fine-grained Text-to-Image Retrieval}, author = {Zehong Ma and Hao Chen and Wei Zeng and Limin Su and Shiliang Zhang}, year = {2025}, doi = {10.1109/tmm.2025.3543066}, url = {https://arxiv.org/abs/2504.07718}, note = {Source identifier: 2504.07718} }