@misc{indiciae81076712f91e, title = {Reinforcement Learning from Diverse Human Preferences}, author = {Wanqi Xue and Bo An and Shuicheng Yan and Zhongwen Xu}, year = {2024}, url = {https://arxiv.org/abs/2301.11774}, note = {Source identifier: 2301.11774} }