@misc{indiciaef658f2e3e1fe, title = {Fast and Efficient 2-bit LLM Inference on GPU: 2/4/16-bit in a Weight Matrix with Asynchronous Dequantization}, author = {Jinhao Li and Jiaming Xu and Shiyao Li and Shan Huang and Jun Liu and Yaoxiu Lian and Guohao Dai}, year = {2024}, url = {https://arxiv.org/abs/2311.16442}, note = {Source identifier: 2311.16442} }