@misc{indiciaea2ecc0cb03da, title = {Ragged Paged Attention: A High-Performance and Flexible LLM Inference Kernel for TPU}, author = {Jevin Jiang and Ying Chen and Blake A. Hechtman and Fenghui Zhang and Yarong Mu}, year = {2026}, url = {https://arxiv.org/abs/2604.15464}, note = {Source identifier: 2604.15464} }