@misc{indiciae7df3b73e5e0c, title = {CLONE: Customizing LLMs for Efficient Latency-Aware Inference at the Edge}, author = {Chunlin Tian and Xinpeng Qin and Kahou Tam and Li Li and Zijian Wang and Yuanzhe Zhao and Minglei Zhang and Chengzhong Xu}, year = {2025}, url = {https://arxiv.org/abs/2506.02847}, note = {Source identifier: 2506.02847} }