@misc{indiciaef9fb7d1b9024, title = {Provable Benefits of Policy Learning from Human Preferences in Contextual Bandit Problems}, author = {Xiang Ji and Huazheng Wang and Minshuo Chen and Tuo Zhao and Mengdi Wang}, year = {2023}, url = {https://arxiv.org/abs/2307.12975}, note = {Source identifier: 2307.12975} }