@misc{indiciaee1905e858780, title = {Agentic Reward Modeling: Integrating Human Preferences with Verifiable Correctness Signals for Reliable Reward Systems}, author = {Hao Peng and Yunjia Qi and Xiaozhi Wang and Zijun Yao and Bin Xu and Lei Hou and Juanzi Li}, year = {2025}, url = {https://arxiv.org/abs/2502.19328}, note = {Source identifier: 2502.19328} }