@misc{indiciae883c7c119c14, title = {Listening to the Echo: User-Reaction Aware Policy Optimization via Scalar-Verbal Hybrid Reinforcement Learning}, author = {Jing Ye and Xinpei Zhao and Lu Xiang and Yaping Zhang and Chengqing Zong}, year = {2026}, url = {https://arxiv.org/abs/2603.15434}, note = {Source identifier: 2603.15434} }