@misc{indiciae03bcb4137dbc, title = {DiTTo-TTS: Diffusion Transformers for Scalable Text-to-Speech without Domain-Specific Factors}, author = {Keon Lee and Dong Won Kim and Jaehyeon Kim and Seungjun Chung and Jaewoong Cho}, year = {2025}, url = {https://arxiv.org/abs/2406.11427}, note = {Source identifier: 2406.11427} }