@misc{indiciaed8578738745a, title = {Llamas on the Web: Memory-Efficient, Performance-Portable, and Multi-Precision LLM Inference with WebGPU}, author = {Reese Levine and Rithik Sharma and Nikhil Jain and Abhijit Ramesh and Zheyuan Chen and Neha Abbas and James Contini and Tyler Sorensen}, year = {2026}, url = {https://arxiv.org/abs/2605.20706}, note = {Source identifier: 2605.20706} }