@article{costefficientllmservinginthecloudvm, title = {Cost-Efficient LLM Serving in the Cloud: VM Selection with KV Cache Offloading}, author = {Kihyun Kim and Jinwoo Kim and Hyunsun Chung and Myung-Hoon Cha and Hong-Yeon Kim and Youngjae Kim}, year = {2025}, eprint = {2504.11816}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2504.11816v1}, }