@misc{indiciae7cf727d12b81, title = {SampleAttention: Near-Lossless Acceleration of Long Context LLM Inference with Adaptive Structured Sparse Attention}, author = {Qianchao Zhu and Jiangfei Duan and Chang Chen and Siran Liu and Guanyu Feng and Xin Lv and Xiao Chuanfu and Dahua Lin and Chao Yang}, year = {2025}, url = {https://arxiv.org/abs/2406.15486}, note = {Source identifier: 2406.15486} }