@misc{indiciae63c198ef95d5, title = {Aligning Information Capacity Between Vision and Language via Dense-to-Sparse Feature Distillation for Image-Text Matching}, author = {Yang Liu and Wentao Feng and Zhuoyao Liu and Shudong Huang and Jiancheng Lv}, year = {2025}, url = {https://arxiv.org/abs/2503.14953}, note = {Source identifier: 2503.14953} }