@misc{indiciae436fd4b25185, title = {Rank-GRPO: Training LLM-based Conversational Recommender Systems with Reinforcement Learning}, author = {Yaochen Zhu and Harald Steck and Dawen Liang and Yinhan He and Vito Ostuni and Jundong Li and Nathan Kallus}, year = {2026}, url = {https://arxiv.org/abs/2510.20150}, note = {Source identifier: 2510.20150} }