@misc{indiciae491299fa18a0, title = {Vision-Language Binding in In-Context Image Generation}, author = {Chris Ge and Rohit Gandikota and Antonio Torralba and Tamar Rott Shaham}, year = {2026}, url = {https://arxiv.org/abs/2605.24624}, note = {Source identifier: 2605.24624} }