@misc{indiciaef9e7aca614cf, title = {RocketKV: Accelerating Long-Context LLM Inference via Two-Stage KV Cache Compression}, author = {Payman Behnam and Yaosheng Fu and Ritchie Zhao and Po-An Tsai and Zhiding Yu and Alexey Tumanov}, year = {2025}, url = {https://arxiv.org/abs/2502.14051}, note = {Source identifier: 2502.14051} }