@article{pelmpowerefficientondevicellminferencewi, title = {PELM: Power Efficient On-Device LLM Inference with Speculative Decoding and Dynamic Voltage Frequency Scaling}, author = {Weisi Yang and Stephen Xia}, year = {2026}, eprint = {2609.09662}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2609.09662}, }