@misc{indiciae29d6e229330d, title = {Compress the Cache, Not the Speech Embedding: KV Compression for Efficient Speech LLMs}, author = {Ke-Han Lu and Keqi Deng and Ruchao Fan and Rui Zhao and Jinyu Li}, year = {2026}, url = {https://arxiv.org/abs/2607.06827}, note = {Source identifier: 2607.06827} }