@misc{indiciae602b82452e23, title = {Seeing Through Words, Speaking Through Pixels: Deep Representational Alignment Between Vision and Language Models}, author = {Zoe Wanying He and Sean Trott and Meenakshi Khosla}, year = {2025}, url = {https://arxiv.org/abs/2509.20751}, note = {Source identifier: 2509.20751} }