@misc{indiciae4a9ad89bd66e, title = {Leftover Lunch: Advantage-based Offline Reinforcement Learning for Language Models}, author = {Ashutosh Baheti and Ximing Lu and Faeze Brahman and Ronan Le Bras and Maarten Sap and Mark Riedl}, year = {2024}, url = {https://arxiv.org/abs/2305.14718}, note = {Source identifier: 2305.14718} }