@misc{indiciae7fb43f2ae364, title = {HPU: High-Bandwidth Processing Unit for Scalable, Cost-effective LLM Inference via GPU Co-processing}, author = {Myunghyun Rhee and Joonseop Sim and Taeyoung Ahn and Seungyong Lee and Daegun Yoon and Euiseok Kim and Kyoung Park and Youngpyo Joo and Hoshik Kim}, year = {2025}, url = {https://arxiv.org/abs/2504.16112}, note = {Source identifier: 2504.16112} }