@misc{indiciae74c4ff1cacce, title = {ScoutAttention: Efficient KV Cache Offloading via Layer-Ahead CPU Pre-computation for LLM Inference}, author = {Qiuyang Zhang and Kai Zhou and Ding Tang and Kai Lu and Cheng Li and Zhenyu Yang and Peng Xu and Jiguang Wan}, year = {2026}, url = {https://arxiv.org/abs/2603.27138}, note = {Source identifier: 2603.27138} }