@misc{indiciae5a7d95f479ec, title = {Improving Reward Models with Synthetic Critiques}, author = {Zihuiwen Ye and Fraser Greenlee-Scott and Max Bartolo and Phil Blunsom and Jon Ander Campos and Matthias Gallé}, year = {2024}, url = {https://arxiv.org/abs/2405.20850}, note = {Source identifier: 2405.20850} }