@misc{indiciaea3c4efb77cbc, title = {Attention-aware Inference Optimizations for Large Vision-Language Models with Memory-efficient Decoding}, author = {Fatih Ilhan and Gaowen Liu and Ramana Rao Kompella and Selim Furkan Tekin and Tiansheng Huang and Zachary Yahn and Yichang Xu and Ling Liu}, year = {2026}, url = {https://arxiv.org/abs/2603.23914}, note = {Source identifier: 2603.23914} }