@misc{indiciae3cc1bbfc1c08, title = {FlashInfer: Efficient and Customizable Attention Engine for LLM Inference Serving}, author = {Zihao Ye and Lequn Chen and Ruihang Lai and Wuwei Lin and Yineng Zhang and Stephanie Wang and Tianqi Chen and Baris Kasikci and Vinod Grover and Arvind Krishnamurthy and Luis Ceze}, year = {2025}, url = {https://arxiv.org/abs/2501.01005}, note = {Source identifier: 2501.01005} }