@misc{indiciae3e84099e2aef, title = {Caption Anything: Interactive Image Description with Diverse Multimodal Controls}, author = {Teng Wang and Jinrui Zhang and Junjie Fei and Hao Zheng and Yunlong Tang and Zhe Li and Mingqi Gao and Shanshan Zhao}, year = {2023}, url = {https://arxiv.org/abs/2305.02677}, note = {Source identifier: 2305.02677} }