@misc{indiciaee9bba6f5d1b5, title = {SqueezeAttention: 2D Management of KV-Cache in LLM Inference via Layer-wise Optimal Budget}, author = {Zihao Wang and Bin Cui and Shaoduo Gan}, year = {2024}, url = {https://arxiv.org/abs/2404.04793}, note = {Source identifier: 2404.04793} }