@misc{indiciae2eb2853d7dea, title = {TokenFlow: Rethinking Fine-grained Cross-modal Alignment in Vision-Language Retrieval}, author = {Xiaohan Zou and Changqiao Wu and Lele Cheng and Zhongyuan Wang}, year = {2022}, url = {https://arxiv.org/abs/2209.13822}, note = {Source identifier: 2209.13822} }