@misc{indiciaef7d03aac708c, title = {Flash-LLM: Enabling Cost-Effective and Highly-Efficient Large Generative Model Inference with Unstructured Sparsity}, author = {Haojun Xia and Zhen Zheng and Yuchao Li and Donglin Zhuang and Zhongzhu Zhou and Xiafei Qiu and Yong Li and Wei Lin and Shuaiwen Leon Song}, year = {2023}, url = {https://arxiv.org/abs/2309.10285}, note = {Source identifier: 2309.10285} }