@misc{indiciae877487e094ba, title = {Audio-conditioned phonemic and prosodic annotation for building text-to-speech models from unlabeled speech data}, author = {Yuma Shirahata and Byeongseon Park and Ryuichi Yamamoto and Kentaro Tachibana}, year = {2024}, url = {https://arxiv.org/abs/2406.08111}, note = {Source identifier: 2406.08111} }