@misc{indiciae72ba81c46afd, title = {SpecPipe: Accelerating Pipeline Parallelism-based LLM Inference with Speculative Decoding}, author = {Haofei Yin and Mengbai Xiao and Tinghong Li and Xiao Zhang and Dongxiao Yu and Guanghui Zhang}, year = {2025}, url = {https://arxiv.org/abs/2504.04104}, note = {Source identifier: 2504.04104} }