@misc{indiciae191066dc28f4, title = {PAT: Accelerating LLM Decoding via Prefix-Aware Attention with Resource Efficient Multi-Tile Kernel}, author = {Jinjun Yi and Zhixin Zhao and Yitao Hu and Ke Yan and Weiwei Sun and Hao Wang and Laiping Zhao and Yuhao Zhang and Wenxin Li and Keqiu Li}, year = {2026}, doi = {10.1145/3779212.3790200}, url = {https://arxiv.org/abs/2511.22333}, note = {Source identifier: 2511.22333} }