@misc{indiciae79661c4baac5, title = {Learning Optimal Advantage from Preferences and Mistaking it for Reward}, author = {W. Bradley Knox and Stephane Hatgis-Kessell and Sigurdur Orn Adalgeirsson and Serena Booth and Anca Dragan and Peter Stone and Scott Niekum}, year = {2023}, url = {https://arxiv.org/abs/2310.02456}, note = {Source identifier: 2310.02456} }