@misc{indiciaef80d568e73ca, title = {InnerQ: Hardware-Aware Tuning-Free Quantization of KV Cache for Large Language Models}, author = {Sayed Mohammadreza Tayaranian Hosseini and Amir Ardakani and Warren J. Gross}, year = {2026}, url = {https://arxiv.org/abs/2602.23200}, note = {Source identifier: 2602.23200} }