@misc{indiciae87c3cd2bc7fb, title = {Efficient Edge LLMs Deployment via HessianAware Quantization and CPU GPU Collaborative}, author = {Tuo Zhang and Ning Li and Xin Yuan and Wenchao Xu and Quan Chen and Song Guo and Haijun Zhang}, year = {2025}, url = {https://arxiv.org/abs/2508.07329}, note = {Source identifier: 2508.07329} }