@article{unlockingefficiencyinlargelanguagemodel, title = {Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding}, author = {Heming Xia and Zhe Yang and Qingxiu Dong and Peiyi Wang and Yongqi Li and Tao Ge and Tianyu Liu and Wenjie Li and Zhifang Sui}, year = {2024}, eprint = {2401.07851}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2401.07851v3}, }