@misc{indiciae6c384deffe48, title = {Sequence to Sequence Reward Modeling: Improving RLHF by Language Feedback}, author = {Jiayi Zhou and Jiaming Ji and Juntao Dai and Dong Li and Yaodong Yang}, year = {2025}, url = {https://arxiv.org/abs/2409.00162}, note = {Source identifier: 2409.00162} }