@misc{indiciae3342ea471147, title = {Tangram: Accelerating Serverless LLM Loading through GPU Memory Reuse and Affinity}, author = {Wenbin Zhu and Zhaoyan Shen and Zili Shao and Hongjun Dai and Feng Chen}, year = {2025}, url = {https://arxiv.org/abs/2512.01357}, note = {Source identifier: 2512.01357} }