@misc{indiciae66748ca7517e, title = {MorphServe: Efficient and Workload-Aware LLM Serving via Runtime Quantized Layer Swapping and KV Cache Resizing}, author = {Zhaoyuan Su and Zeyu Zhang and Tingfeng Lan and Zirui Wang and Haiying Shen and Juncheng Yang and Yue Cheng}, year = {2026}, url = {https://arxiv.org/abs/2506.02006}, note = {Source identifier: 2506.02006} }