@misc{indiciae1c45416e8147, title = {f-GRPO and Beyond: Divergence-Based Reinforcement Learning Algorithms for General LLM Alignment}, author = {Rajdeep Haldar and Lantao Mei and Guang Lin and Yue Xing and Qifan Song}, year = {2026}, url = {https://arxiv.org/abs/2602.05946}, note = {Source identifier: 2602.05946} }