@misc{indiciae83231d02198b, title = {Get More with LESS: Synthesizing Recurrence with KV Cache Compression for Efficient LLM Inference}, author = {Harry Dong and Xinyu Yang and Zhenyu Zhang and Zhangyang Wang and Yuejie Chi and Beidi Chen}, year = {2024}, url = {https://arxiv.org/abs/2402.09398}, note = {Source identifier: 2402.09398} }