@misc{indiciaefa7ad1efaa72, title = {Robust Reinforcement Learning from Corrupted Human Feedback}, author = {Alexander Bukharin and Ilgee Hong and Haoming Jiang and Zichong Li and Qingru Zhang and Zixuan Zhang and Tuo Zhao}, year = {2024}, url = {https://arxiv.org/abs/2406.15568}, note = {Source identifier: 2406.15568} }