@misc{indiciae89d9ced67608, title = {Noise-robust zero-shot text-to-speech synthesis conditioned on self-supervised speech-representation model with adapters}, author = {Kenichi Fujita and Hiroshi Sato and Takanori Ashihara and Hiroki Kanagawa and Marc Delcroix and Takafumi Moriya and Yusuke Ijima}, year = {2024}, url = {https://arxiv.org/abs/2401.05111}, note = {Source identifier: 2401.05111} }