@misc{indiciaea4816d8da432, title = {OBCache: Optimal Brain KV Cache Pruning for Efficient Long-Context LLM Inference}, author = {Yuzhe Gu and Xiyu Liang and Jiaojiao Zhao and Enmao Diao}, year = {2026}, url = {https://arxiv.org/abs/2510.07651}, note = {Source identifier: 2510.07651} }