@misc{indiciae6c9a58b33fdb, title = {98\$\textbackslash{}times\$ Faster LLM Routing Without a Dedicated GPU: Flash Attention, Prompt Compression, and Near-Streaming for the vLLM Semantic Router}, author = {Xunzhuo Liu and Bowei He and Xue Liu and Andy Luo and Haichen Zhang and Huamin Chen}, year = {2026}, url = {https://arxiv.org/abs/2603.12646}, note = {Source identifier: 2603.12646} }