@misc{indiciae7de68919e46f, title = {On-Device Qwen2.5: Efficient LLM Inference with Model Compression and Hardware Acceleration}, author = {Maoyang Xiang and Ramesh Fernando and Bo Wang}, year = {2025}, url = {https://arxiv.org/abs/2504.17376}, note = {Source identifier: 2504.17376} }