@misc{indiciae56c19e8e48fb, title = {Emphasis Rendering for Conversational Text-to-Speech with Multi-modal Multi-scale Context Modeling}, author = {Rui Liu and Zhenqi Jia and Jie Yang and Yifan Hu and Haizhou Li}, year = {2024}, url = {https://arxiv.org/abs/2410.09524}, note = {Source identifier: 2410.09524} }