@misc{indiciaea521441b95af, title = {WKVQuant: Quantizing Weight and Key/Value Cache for Large Language Models Gains More}, author = {Yuxuan Yue and Zhihang Yuan and Haojie Duanmu and Sifan Zhou and Jianlong Wu and Liqiang Nie}, year = {2024}, url = {https://arxiv.org/abs/2402.12065}, note = {Source identifier: 2402.12065} }