@article{preserveprefetchingmodelweightsandkv, title = {PRESERVE: Prefetching Model Weights and KV-Cache in Distributed LLM Serving}, author = {Ahmet Caner Yüzügüler and Jiawei Zhuang and Lukas Cavigelli}, year = {2025}, eprint = {2501.08192}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2501.08192v2}, }