@misc{indiciae7737efcb7f61, title = {Sparser is Faster and Less is More: Efficient Sparse Attention for Long-Range Transformers}, author = {Chao Lou and Zixia Jia and Zilong Zheng and Kewei Tu}, year = {2024}, url = {https://arxiv.org/abs/2406.16747}, note = {Source identifier: 2406.16747} }