@misc{indiciaede2020cc74c8, title = {Layer-Condensed KV Cache for Efficient Inference of Large Language Models}, author = {Haoyi Wu and Kewei Tu}, year = {2024}, url = {https://arxiv.org/abs/2405.10637}, note = {Source identifier: 2405.10637} }