@misc{indiciae15066c89d584, title = {Improving Audio Codec-based Zero-Shot Text-to-Speech Synthesis with Multi-Modal Context and Large Language Model}, author = {Jinlong Xue and Yayue Deng and Yicheng Han and Yingming Gao and Ya Li}, year = {2024}, url = {https://arxiv.org/abs/2406.03706}, note = {Source identifier: 2406.03706} }