@misc{indiciae80ef2a7702c7, title = {VoxInstruct: Expressive Human Instruction-to-Speech Generation with Unified Multilingual Codec Language Modelling}, author = {Yixuan Zhou and Xiaoyu Qin and Zeyu Jin and Shuoyi Zhou and Shun Lei and Songtao Zhou and Zhiyong Wu and Jia Jia}, year = {2024}, doi = {10.1145/3664647.3681680}, url = {https://arxiv.org/abs/2408.15676}, note = {Source identifier: 2408.15676} }