@article{hpuhighbandwidthprocessingunitfor, title = {HPU: High-Bandwidth Processing Unit for Scalable, Cost-effective LLM Inference via GPU Co-processing}, author = {Myunghyun Rhee and Joonseop Sim and Taeyoung Ahn and Seungyong Lee and Daegun Yoon and Euiseok Kim and Kyoung Park and Youngpyo Joo and Hosik Kim}, year = {2025}, eprint = {2504.16112}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2504.16112v1}, }