@misc{indiciae0b08aeddea6d, title = {Accelerating LLM Inference Throughput via Asynchronous KV Cache Prefetching}, author = {Yanhao Dong and Yubo Miao and Weinan Li and Xiao Zheng and Chao Wang and Jiesheng Wu and Feng Lyu}, year = {2025}, url = {https://arxiv.org/abs/2504.06319}, note = {Source identifier: 2504.06319} }