@misc{indiciae0d8734ef60fe, title = {Ada-KV: Optimizing KV Cache Eviction by Adaptive Budget Allocation for Efficient LLM Inference}, author = {Yuan Feng and Junlin Lv and Yukun Cao and Xike Xie and S. Kevin Zhou}, year = {2025}, url = {https://arxiv.org/abs/2407.11550}, note = {Source identifier: 2407.11550} }