@article{scalableprocessingnearmemoryfor1mtokenll, title = {Scalable Processing-Near-Memory for 1M-Token LLM Inference: CXL-Enabled KV-Cache Management Beyond GPU Limits}, author = {Dowon Kim and MinJae Lee and Janghyeon Kim and HyuckSung Kwon and Hyeonggyu Jeong and Sang-Soo Park and Minyong Yoon and Si-Dong Roh and Yongsuk Kwon and Jinin So and Jungwook Choi}, year = {2025}, eprint = {2511.00321}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2511.00321}, }