@misc{indiciae3a3c9c1264d7, title = {LLaMCAT: Optimizing Large Language Model Inference with Cache Arbitration and Throttling}, author = {Zhongchun Zhou and Chengtao Lai and Wei Zhang}, year = {2025}, doi = {10.1145/3754598.3754671}, url = {https://arxiv.org/abs/2512.00083}, note = {Source identifier: 2512.00083} }