@misc{indiciaee36b606572af, title = {LLMCache: Layer-Wise Caching Strategies for Accelerated Reuse in Transformer Inference}, author = {Harsh Vardhan Bansal}, year = {2025}, doi = {10.1109/ised67359.2025.11405274}, url = {https://arxiv.org/abs/2512.16843}, note = {Source identifier: 2512.16843} }