@article{nuqmmquantizedmatmulforefficient, title = {LUT-GEMM: Quantized Matrix Multiplication based on LUTs for Efficient Inference in Large-Scale Generative Language Models}, author = {Gunho Park and Baeseong Park and Minsub Kim and Sungjae Lee and Jeonghoon Kim and Beomseok Kwon and Se Jung Kwon and Byeongwook Kim and Youngjoo Lee and Dongsoo Lee}, year = {2022}, eprint = {2206.09557}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2206.09557v4}, }