@article{improvingtheservingperformanceofmulti, title = {Improving the Serving Performance of Multi-LoRA Large Language Models via Efficient LoRA and KV Cache Management}, author = {Hang Zhang and Jiuchen Shi and Yixiao Wang and Quan Chen and Yizhou Shan and Minyi Guo}, year = {2025}, eprint = {2505.03756}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2505.03756v1}, }