@misc{indiciae0bf47ba01614, title = {MELA-TTS: Joint transformer-diffusion model with representation alignment for speech synthesis}, author = {Keyu An and Zhiyu Zhang and Changfeng Gao and Yabin Li and Zhendong Peng and Haoxu Wang and Zhihao Du and Han Zhao and Zhifu Gao and Xiangang Li}, year = {2026}, url = {https://arxiv.org/abs/2509.14784}, note = {Source identifier: 2509.14784} }