@misc{indiciaeb2e08834e47d, title = {KV Cache is 1 Bit Per Channel: Efficient Large Language Model Inference with Coupled Quantization}, author = {Tianyi Zhang and Jonah Yi and Zhaozhuo Xu and Anshumali Shrivastava}, year = {2024}, url = {https://arxiv.org/abs/2405.03917}, note = {Source identifier: 2405.03917} }