@article{moskamixtureofsharedkvattentionforeffici, title = {MoSKA: Mixture of Shared KV Attention for Efficient Long-Sequence LLM Inference}, author = {Myunghyun Rhee and Sookyung Choi and Euiseok Kim and Joonseop Sim and Youngpyo Joo and Hoshik Kim}, year = {2025}, eprint = {2511.06010}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2511.06010}, }