@misc{indiciaea3ec9aa0d8f9, title = {Utilizing Neural Transducers for Two-Stage Text-to-Speech via Semantic Token Prediction}, author = {Minchan Kim and Myeonghun Jeong and Byoung Jin Choi and Semin Kim and Joun Yeop Lee and Nam Soo Kim}, year = {2024}, url = {https://arxiv.org/abs/2401.01498}, note = {Source identifier: 2401.01498} }