@misc{indiciae92b5cb27726f, title = {CARD: A Cache-Assisted Parallel Speculative Decoding Framework via Query-and-Correct Paradigm for Accelerating LLM Inference}, author = {Enyu Zhou and Kai Sheng and Hao Chen and Xin He}, year = {2025}, url = {https://arxiv.org/abs/2508.04462}, note = {Source identifier: 2508.04462} }