@misc{indiciaea94b4ceadab4, title = {KVTuner: Sensitivity-Aware Layer-Wise Mixed-Precision KV Cache Quantization for Efficient and Nearly Lossless LLM Inference}, author = {Xing Li and Zeyu Xing and Yiming Li and Linping Qu and Hui-Ling Zhen and Wulong Liu and Yiwu Yao and Sinno Jialin Pan and Mingxuan Yuan}, year = {2025}, url = {https://arxiv.org/abs/2502.04420}, note = {Source identifier: 2502.04420} }