@misc{indiciaef5e943373371, title = {Batch-Max: Higher LLM Throughput using Larger Batch Sizes and KV Cache Compression}, author = {Michael R. Metel and Boxing Chen and Mehdi Rezagholizadeh}, year = {2025}, url = {https://arxiv.org/abs/2412.05693}, note = {Source identifier: 2412.05693} }