@misc{indiciaebfca51e6a96c, title = {GEAR: An Efficient KV Cache Compression Recipe for Near-Lossless Generative Inference of LLM}, author = {Hao Kang and Qingru Zhang and Souvik Kundu and Geonhwa Jeong and Zaoxing Liu and Tushar Krishna and Tuo Zhao}, year = {2024}, url = {https://arxiv.org/abs/2403.05527}, note = {Source identifier: 2403.05527} }