@misc{indiciae3e9b5c7c3d4b, title = {PRESERVE: Prefetching Model Weights and KV-Cache in Distributed LLM Serving}, author = {Ahmet Caner Yüzügüler and Jiawei Zhuang and Lukas Cavigelli}, year = {2025}, url = {https://arxiv.org/abs/2501.08192}, note = {Source identifier: 2501.08192} }