@misc{indiciaed2f2b1df96cc, title = {CompressKV: Semantic-Retrieval-Guided KV-Cache Compression for Resource-Efficient Long-Context LLM Inference}, author = {Xiaolin Lin and Jingcun Wang and Olga Kondrateva and Yiyu Shi and Bing Li and Grace Li Zhang}, year = {2026}, url = {https://arxiv.org/abs/2606.24467}, note = {Source identifier: 2606.24467} }