@misc{indiciae7619e1b696e7, title = {The Limits of Learning from Pictures and Text: Vision-Language Models and Embodied Scene Understanding}, author = {Gillian Rosenberg and Skylar Stadhard and Bruce C. Hansen and Michelle R. Greene}, year = {2026}, url = {https://arxiv.org/abs/2603.26589}, note = {Source identifier: 2603.26589} }