@misc{indiciae04096d019ee3, title = {LLM Inference at the Edge: Mobile, NPU, and GPU Performance Efficiency Trade-offs Under Sustained Load}, author = {Pranay Tummalapalli and Sahil Arayakandy and Ritam Pal and Kautuk Kundan}, year = {2026}, url = {https://arxiv.org/abs/2603.23640}, note = {Source identifier: 2603.23640} }