@misc{indiciaeee356ba7c2bf, title = {Countering Reward Over-optimization in LLM with Demonstration-Guided Reinforcement Learning}, author = {Mathieu Rita and Florian Strub and Rahma Chaabouni and Paul Michel and Emmanuel Dupoux and Olivier Pietquin}, year = {2024}, url = {https://arxiv.org/abs/2404.19409}, note = {Source identifier: 2404.19409} }