@misc{indiciae5b73fd5dff6f, title = {Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve}, author = {Amey Agrawal and Nitin Kedia and Ashish Panwar and Jayashree Mohan and Nipun Kwatra and Bhargav S. Gulavani and Alexey Tumanov and Ramachandran Ramjee}, year = {2024}, url = {https://arxiv.org/abs/2403.02310}, note = {Source identifier: 2403.02310} }