@misc{indiciaee5e8136274e3, title = {Multimodal Representation Alignment for Image Generation: Text-Image Interleaved Control Is Easier Than You Think}, author = {Liang Chen and Shuai Bai and Wenhao Chai and Weichu Xie and Haozhe Zhao and Leon Vinci and Junyang Lin and Baobao Chang}, year = {2025}, url = {https://arxiv.org/abs/2502.20172}, note = {Source identifier: 2502.20172} }