@misc{indiciae5d8a257a2cbb, title = {ZET-Speech: Zero-shot adaptive Emotion-controllable Text-to-Speech Synthesis with Diffusion and Style-based Models}, author = {Minki Kang and Wooseok Han and Sung Ju Hwang and Eunho Yang}, year = {2023}, url = {https://arxiv.org/abs/2305.13831}, note = {Source identifier: 2305.13831} }