@article{efficientarbitraryprecisionacceleration, title = {Efficient Arbitrary Precision Acceleration for Large Language Models on GPU Tensor Cores}, author = {Shaobo Ma and Chao Fang and Haikuo Shao and Zhongfeng Wang}, year = {2024}, eprint = {2409.17870}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2409.17870v2}, }