@misc{indiciae915eb20c6eee, title = {RS-DPO: A Hybrid Rejection Sampling and Direct Preference Optimization Method for Alignment of Large Language Models}, author = {Saeed Khaki and JinJin Li and Lan Ma and Liu Yang and Prathap Ramachandra}, year = {2024}, url = {https://arxiv.org/abs/2402.10038}, note = {Source identifier: 2402.10038} }