@misc{indiciae199698860256, title = {Step-Controlled DPO: Leveraging Stepwise Error for Enhanced Mathematical Reasoning}, author = {Zimu Lu and Aojun Zhou and Ke Wang and Houxing Ren and Weikang Shi and Junting Pan and Mingjie Zhan and Hongsheng Li}, year = {2024}, url = {https://arxiv.org/abs/2407.00782}, note = {Source identifier: 2407.00782} }