@article{progressivemixedprecisiondecodingfor, title = {Progressive Mixed-Precision Decoding for Efficient LLM Inference}, author = {Hao Mark Chen and Fuwen Tan and Alexandros Kouris and Royson Lee and Hongxiang Fan and Stylianos I. Venieris}, year = {2024}, eprint = {2410.13461}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2410.13461v1}, }