@misc{indiciaea074c75effc5, title = {Learning Ordinal Probabilistic Reward from Preferences}, author = {Longze Chen and Lu Wang and Renke Shan and Ze Gong and Run Luo and Jiaming Li and Jing Luo and Qiyao Wang and Min Yang}, year = {2026}, url = {https://arxiv.org/abs/2602.12660}, note = {Source identifier: 2602.12660} }