@misc{indiciae1de80b005757, title = {Strategyproof Reinforcement Learning from Human Feedback}, author = {Thomas Kleine Buening and Jiarui Gan and Debmalya Mandal and Marta Kwiatkowska}, year = {2025}, url = {https://arxiv.org/abs/2503.09561}, note = {Source identifier: 2503.09561} }