@misc{indiciae13f89aa16cfb, title = {Why do LLaVA Vision-Language Models Reply to Images in English?}, author = {Musashi Hinck and Carolin Holtermann and Matthew Lyle Olson and Florian Schneider and Sungduk Yu and Anahita Bhiwandiwalla and Anne Lauscher and Shaoyen Tseng and Vasudev Lal}, year = {2024}, url = {https://arxiv.org/abs/2407.02333}, note = {Source identifier: 2407.02333} }