@misc{indiciaec7f489fd1009, title = {Sample-Efficient Reinforcement Learning from Human Feedback via Information-Directed Sampling}, author = {Han Qi and Haochen Yang and Qiaosheng Zhang and Zhuoran Yang}, year = {2025}, url = {https://arxiv.org/abs/2502.05434}, note = {Source identifier: 2502.05434} }