@misc{indiciae8ae46f17ab56, title = {StyleFusion TTS: Multimodal Style-control and Enhanced Feature Fusion for Zero-shot Text-to-speech Synthesis}, author = {Zhiyong Chen and Xinnuo Li and Zhiqi Ai and Shugong Xu}, year = {2024}, url = {https://arxiv.org/abs/2409.15741}, note = {Source identifier: 2409.15741} }