@misc{indiciaeb7a4fbfe4d44, title = {Mastering Text-to-Image Diffusion: Recaptioning, Planning, and Generating with Multimodal LLMs}, author = {Ling Yang and Zhaochen Yu and Chenlin Meng and Minkai Xu and Stefano Ermon and Bin Cui}, year = {2024}, url = {https://arxiv.org/abs/2401.11708}, note = {Source identifier: 2401.11708} }