@misc{indiciae0ef5f4d0ca9a, title = {CoCa: Contrastive Captioners are Image-Text Foundation Models}, author = {Jiahui Yu and Zirui Wang and Vijay Vasudevan and Legg Yeung and Mojtaba Seyedhosseini and Yonghui Wu}, year = {2022}, url = {https://arxiv.org/abs/2205.01917}, note = {Source identifier: 2205.01917} }