@misc{indiciaec0a65f3190ee, title = {Addax: Utilizing Zeroth-Order Gradients to Improve Memory Efficiency and Performance of SGD for Fine-Tuning Language Models}, author = {Zeman Li and Xinwei Zhang and Peilin Zhong and Yuan Deng and Meisam Razaviyayn and Vahab Mirrokni}, year = {2024}, url = {https://arxiv.org/abs/2410.06441}, note = {Source identifier: 2410.06441} }