@article{gptqtquantizelargelanguagemodelstwiceto, title = {GPTQT: Quantize Large Language Models Twice to Push the Efficiency}, author = {Yipin Guo and Yilin Lang and Qinyuan Ren}, year = {2024}, eprint = {2407.02891}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2407.02891v1}, }