@article{moamixtureofsparseattentionforautomatic, title = {MoA: Mixture of Sparse Attention for Automatic Large Language Model Compression}, author = {Tianyu Fu and Haofeng Huang and Xuefei Ning and Genghan Zhang and Boju Chen and Tianqi Wu and Hongyi Wang and Zixiao Huang and Shiyao Li and Shengen Yan and Guohao Dai and Huazhong Yang and Yu Wang}, year = {2024}, eprint = {2406.14909}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2406.14909v2}, }