@misc{indiciae089e54d5b12d, title = {Controllable Text-to-Speech Synthesis with Masked-Autoencoded Style-Rich Representation}, author = {Yongqi Wang and Chunlei Zhang and Hangting Chen and Zhou Zhao and Dong Yu}, year = {2025}, url = {https://arxiv.org/abs/2506.02997}, note = {Source identifier: 2506.02997} }