@misc{indiciaeec134a3ca46b, title = {BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation}, author = {Junnan Li and Dongxu Li and Caiming Xiong and Steven Hoi}, year = {2022}, url = {https://arxiv.org/abs/2201.12086}, note = {Source identifier: 2201.12086} }