@misc{indiciae4781a5624e1d, title = {Trust Region Preference Approximation: A simple and stable reinforcement learning algorithm for LLM reasoning}, author = {Xuerui Su and Shufang Xie and Guoqing Liu and Yingce Xia and Renqian Luo and Peiran Jin and Zhiming Ma and Yue Wang and Zun Wang and Yuting Liu}, year = {2025}, url = {https://arxiv.org/abs/2504.04524}, note = {Source identifier: 2504.04524} }