@misc{indiciaebcd09f1ef8b8, title = {Multi-Grained Vision Language Pre-Training: Aligning Texts with Visual Concepts}, author = {Yan Zeng and Xinsong Zhang and Hang Li}, year = {2022}, url = {https://arxiv.org/abs/2111.08276}, note = {Source identifier: 2111.08276} }