@misc{indiciaed8c9c246c7e5, title = {From RLHF to Direct Alignment: A Theoretical Unification of Preference Learning for Large Language Models}, author = {Tarun Raheja and Nilay Pochhi}, year = {2026}, url = {https://arxiv.org/abs/2601.06108}, note = {Source identifier: 2601.06108} }