@misc{indiciae3e528382c90c, title = {Zeroth-Order Policy Gradient for Reinforcement Learning from Human Feedback without Reward Inference}, author = {Qining Zhang and Lei Ying}, year = {2025}, url = {https://arxiv.org/abs/2409.17401}, note = {Source identifier: 2409.17401} }