@misc{indiciae333c45602f0a, title = {RLHFPoison: Reward Poisoning Attack for Reinforcement Learning with Human Feedback in Large Language Models}, author = {Jiongxiao Wang and Junlin Wu and Muhao Chen and Yevgeniy Vorobeychik and Chaowei Xiao}, year = {2024}, url = {https://arxiv.org/abs/2311.09641}, note = {Source identifier: 2311.09641} }