@misc{indiciaea14dcf2dc04a, title = {ViT-TTS: Visual Text-to-Speech with Scalable Diffusion Transformer}, author = {Huadai Liu and Rongjie Huang and Xuan Lin and Wenqiang Xu and Maozong Zheng and Hong Chen and Jinzheng He and Zhou Zhao}, year = {2024}, url = {https://arxiv.org/abs/2305.12708}, note = {Source identifier: 2305.12708} }