@misc{indiciaeee1f383ac21a, title = {Cascade: Exploiting SLO-Aware latency budget for fair and high goodput LLM inference serving}, author = {Muhammad Adnan and Rohan Mahapatra and Prashant J. Nair and Daniel Berger and Pantea Zardoshti and Rodrigo Fonseca and Esha Choukse}, year = {2026}, url = {https://arxiv.org/abs/2608.06557}, note = {Source identifier: 2608.06557} }