@misc{indiciaefda6c78015a2, title = {BatchLLM: Optimizing Large Batched LLM Inference with Global Prefix Sharing and Throughput-oriented Token Batching}, author = {Zhen Zheng and Xin Ji and Taosong Fang and Fanghao Zhou and Chuanjie Liu and Gang Peng}, year = {2026}, url = {https://arxiv.org/abs/2412.03594}, note = {Source identifier: 2412.03594} }