@misc{indiciae6f60ae00d5a9, title = {Magnet: We Never Know How Text-to-Image Diffusion Models Work, Until We Learn How Vision-Language Models Function}, author = {Chenyi Zhuang and Ying Hu and Pan Gao}, year = {2024}, url = {https://arxiv.org/abs/2409.19967}, note = {Source identifier: 2409.19967} }