@misc{indiciae756d3b528914, title = {Enhancing Multimodal Large Language Models with Multi-instance Visual Prompt Generator for Visual Representation Enrichment}, author = {Wenliang Zhong and Wenyi Wu and Qi Li and Rob Barton and Boxin Du and Shioulin Sam and Karim Bouyarmane and Ismail Tutar and Junzhou Huang}, year = {2024}, url = {https://arxiv.org/abs/2406.02987}, note = {Source identifier: 2406.02987} }