@misc{indiciaee20f7bda6abb, title = {Uncertainty-Penalized Reinforcement Learning from Human Feedback with Diverse Reward LoRA Ensembles}, author = {Yuanzhao Zhai and Han Zhang and Yu Lei and Yue Yu and Kele Xu and Dawei Feng and Bo Ding and Huaimin Wang}, year = {2023}, url = {https://arxiv.org/abs/2401.00243}, note = {Source identifier: 2401.00243} }