@misc{indiciae6b10fe824439, title = {Online Iterative Reinforcement Learning from Human Feedback with General Preference Model}, author = {Chenlu Ye and Wei Xiong and Yuheng Zhang and Hanze Dong and Nan Jiang and Tong Zhang}, year = {2024}, url = {https://arxiv.org/abs/2402.07314}, note = {Source identifier: 2402.07314} }