@misc{indiciaeff6a934cc016, title = {Utility-inspired Reward Transformations Improve Reinforcement Learning Training of Language Models}, author = {Roberto-Rafael Maura-Rivero and Chirag Nagpal and Roma Patel and Francesco Visin}, year = {2025}, url = {https://arxiv.org/abs/2501.06248}, note = {Source identifier: 2501.06248} }