@misc{indiciae4ee066f877e6, title = {Why Do Speech Language Models Fail to Generate Semantically Coherent Outputs? A Modality Evolving Perspective}, author = {Hankun Wang and Haoran Wang and Yiwei Guo and Zhihan Li and Chenpeng Du and Kai Yu}, year = {2026}, url = {https://arxiv.org/abs/2412.17048}, note = {Source identifier: 2412.17048} }