@misc{indiciae6c6845c91e67, title = {Contextual inference from single objects in Vision-Language models}, author = {Martina G. Vilas and Timothy Schaumlöffel and Gemma Roig}, year = {2026}, url = {https://arxiv.org/abs/2603.26731}, note = {Source identifier: 2603.26731} }