@misc{indiciae3e2932a9f865, title = {Multi-modal Generative AI: Multi-modal LLMs, Diffusions, and the Unification}, author = {Xin Wang and Yuwei Zhou and Bin Huang and Hong Chen and Wenwu Zhu}, year = {2025}, doi = {10.1109/tcsvt.2025.3635224}, url = {https://arxiv.org/abs/2409.14993}, note = {Source identifier: 2409.14993} }