@misc{indiciaea0d99f32668c, title = {ViP-LLaVA: Making Large Multimodal Models Understand Arbitrary Visual Prompts}, author = {Mu Cai and Haotian Liu and Dennis Park and Siva Karthik Mustikovela and Gregory P. Meyer and Yuning Chai and Yong Jae Lee}, year = {2024}, url = {https://arxiv.org/abs/2312.00784}, note = {Source identifier: 2312.00784} }