@misc{indiciae952a614502bc, title = {Vision-Language Models Create Cross-Modal Task Representations}, author = {Grace Luo and Trevor Darrell and Amir Bar}, year = {2025}, url = {https://arxiv.org/abs/2410.22330}, note = {Source identifier: 2410.22330} }