@misc{indiciae6802f60b93e6, title = {Exploring the Deep Fusion of Large Language Models and Diffusion Transformers for Text-to-Image Synthesis}, author = {Bingda Tang and Boyang Zheng and Xichen Pan and Sayak Paul and Saining Xie}, year = {2025}, url = {https://arxiv.org/abs/2505.10046}, note = {Source identifier: 2505.10046} }