@misc{indiciaecff945d04200, title = {Accelerating GPU Inference of Large Language Models with Moderately Unstructured Sparse Weight Matrices}, author = {Tao Lu and Haoyu Wang and Zonghui Wang and Keshen Xiang and Jiaheng Zhang and Wenzhi Chen}, year = {2026}, url = {https://arxiv.org/abs/2607.08786}, note = {Source identifier: 2607.08786} }