@misc{indiciaeeab598f1cffc, title = {VGDiffZero: Text-to-image Diffusion Models Can Be Zero-shot Visual Grounders}, author = {Xuyang Liu and Siteng Huang and Yachen Kang and Honggang Chen and Donglin Wang}, year = {2024}, url = {https://arxiv.org/abs/2309.01141}, note = {Source identifier: 2309.01141} }