@misc{indiciae1cbd472d3362, title = {Preference Poisoning Attacks on Reward Model Learning}, author = {Junlin Wu and Jiongxiao Wang and Chaowei Xiao and Chenguang Wang and Ning Zhang and Yevgeniy Vorobeychik}, year = {2024}, url = {https://arxiv.org/abs/2402.01920}, note = {Source identifier: 2402.01920} }