@misc{indiciaef31562fe6eb7, title = {Teaching Large Language Models to Reason with Reinforcement Learning}, author = {Alex Havrilla and Yuqing Du and Sharath Chandra Raparthy and Christoforos Nalmpantis and Jane Dwivedi-Yu and Maksym Zhuravinskyi and Eric Hambro and Sainbayar Sukhbaatar and Roberta Raileanu}, year = {2024}, url = {https://arxiv.org/abs/2403.04642}, note = {Source identifier: 2403.04642} }