@misc{indiciae151c518eb139, title = {Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding}, author = {Heming Xia and Zhe Yang and Qingxiu Dong and Peiyi Wang and Yongqi Li and Tao Ge and Tianyu Liu and Wenjie Li and Zhifang Sui}, year = {2024}, url = {https://arxiv.org/abs/2401.07851}, note = {Source identifier: 2401.07851} }