@misc{indiciaeac2ffef582fa, title = {An Efficient Hybrid Sparse Attention with CPU-GPU Parallelism for Long-Context Inference}, author = {Feiyu Yao and Zhixiong Niu and Xiaqing Li and Yongqiang Xiong and Juan Fang and Qian Wang}, year = {2026}, url = {https://arxiv.org/abs/2605.07719}, note = {Source identifier: 2605.07719} }