@article{efficientedgellmsdeploymentviahessianawa, title = {Efficient Edge LLMs Deployment via HessianAware Quantization and CPU GPU Collaborative}, author = {Tuo Zhang and Ning Li and Xin Yuan and Wenchao Xu and Quan Chen and Song Guo and Haijun Zhang}, year = {2025}, eprint = {2508.07329}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2508.07329}, }