@misc{indiciaeefa09b571048, title = {Q Cache: Visual Attention is Valuable in Less than Half of Decode Layers for Multimodal Large Language Model}, author = {Jiedong Zhuang and Lu Lu and Ming Dai and Rui Hu and Jian Chen and Qiang Liu and Haoji Hu}, year = {2026}, url = {https://arxiv.org/abs/2602.01901}, note = {Source identifier: 2602.01901} }