@misc{indiciaef7960bae71ac, title = {VoiceTextBlender: Augmenting Large Language Models with Speech Capabilities via Single-Stage Joint Speech-Text Supervised Fine-Tuning}, author = {Yifan Peng and Krishna C. Puvvada and Zhehuai Chen and Piotr Zelasko and He Huang and Kunal Dhawan and Ke Hu and Shinji Watanabe and Jagadeesh Balam and Boris Ginsburg}, year = {2025}, url = {https://arxiv.org/abs/2410.17485}, note = {Source identifier: 2410.17485} }