@misc{indiciaebf4e077a6631, title = {Towards zero-shot Text-based voice editing using acoustic context conditioning, utterance embeddings, and reference encoders}, author = {Jason Fong and Yun Wang and Prabhav Agrawal and Vimal Manohar and Jilong Wu and Thilo Köhler and Qing He}, year = {2022}, url = {https://arxiv.org/abs/2210.16045}, note = {Source identifier: 2210.16045} }