@misc{indiciae18d061aafb72, title = {Hardware-Aware Parallel Prompt Decoding for Memory-Efficient Acceleration of LLM Inference}, author = {Hao Mark Chen and Wayne Luk and Ka Fai Cedric Yiu and Rui Li and Konstantin Mishchenko and Stylianos I. Venieris and Hongxiang Fan}, year = {2025}, url = {https://arxiv.org/abs/2405.18628}, note = {Source identifier: 2405.18628} }