@misc{indiciaeb630c8a5af60, title = {Adapting Text LLMs to Speech via Multimodal Depth Up-Scaling}, author = {Kazuki Yano and Jun Suzuki and Shinji Watanabe}, year = {2026}, url = {https://arxiv.org/abs/2604.00489}, note = {Source identifier: 2604.00489} }