@misc{indiciae166f0e7e1cdf, title = {Full-Step-DPO: Self-Supervised Preference Optimization with Step-wise Rewards for Mathematical Reasoning}, author = {Huimin Xu and Xin Mao and Feng-Lin Li and Xiaobao Wu and Wang Chen and Wei Zhang and Anh Tuan Luu}, year = {2025}, url = {https://arxiv.org/abs/2502.14356}, note = {Source identifier: 2502.14356} }