@misc{indiciae916ba1150b0d, title = {Prompt Cache: Modular Attention Reuse for Low-Latency Inference}, author = {In Gim and Guojun Chen and Seung-seob Lee and Nikhil Sarda and Anurag Khandelwal and Lin Zhong}, year = {2024}, url = {https://arxiv.org/abs/2311.04934}, note = {Source identifier: 2311.04934} }