@misc{indiciaebc8e837a26db, title = {Simultaneous Reward Distillation and Preference Learning: Get You a Language Model Who Can Do Both}, author = {Abhijnan Nath and Changsoo Jung and Ethan Seefried and Nikhil Krishnaswamy}, year = {2025}, url = {https://arxiv.org/abs/2410.08458}, note = {Source identifier: 2410.08458} }