@article{flashmoereducingssdiobottlenecksviamlbas, title = {FlashMoE: Reducing SSD I/O Bottlenecks via ML-Based Cache Replacement for Mixture-of-Experts Inference on Edge Devices}, author = {Byeongju Kim and Jungwan Lee and Donghyeon Han and Hoi-Jun Yoo and Sangyeob Kim}, year = {2026}, eprint = {2601.17063}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2601.17063}, }