@misc{indiciae234ca0455451, title = {How Visual Representations Map to Language Feature Space in Multimodal LLMs}, author = {Constantin Venhoff and Ashkan Khakzar and Sonia Joseph and Philip Torr and Neel Nanda}, year = {2025}, url = {https://arxiv.org/abs/2506.11976}, note = {Source identifier: 2506.11976} }