@misc{indiciae4df8f65113cb, title = {EMMA: Efficient Multimodal Understanding, Generation, and Editing with a Unified Architecture}, author = {Xin He and Longhui Wei and Jianbo Ouyang and Minghui Liao and Lingxi Xie and Qi Tian}, year = {2025}, url = {https://arxiv.org/abs/2512.04810}, note = {Source identifier: 2512.04810} }