@article{puzzlemoeefficientcompressionoflargemixt, title = {PuzzleMoE: Efficient Compression of Large Mixture-of-Experts Models via Sparse Expert Merging and Bit-packed inference}, author = {Yushu Zhao and Zheng Wang and Minjia Zhang}, year = {2025}, eprint = {2511.04805}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2511.04805}, }