@misc{indiciaeb4a7f0075567, title = {SeaLLM: Service-Aware and Latency-Optimized Resource Sharing for Large Language Model Inference}, author = {Yihao Zhao and Jiadun Chen and Peng Sun and Lei Li and Xuanzhe Liu and Xin Jin}, year = {2025}, url = {https://arxiv.org/abs/2504.15720}, note = {Source identifier: 2504.15720} }