@misc{indiciae31d0437dd9bf, title = {Alternating Reinforcement Learning with Contextual Rubric Rewards: Beyond the Scalarization Strategy}, author = {Guangchen Lan and Lian Xiong and Xin Zhou and Hejie Cui and Yuwei Zhang and Mao Li and Zhenyu Shi and Besnik Fetahu and Lihong Li and Xian Li}, year = {2026}, url = {https://arxiv.org/abs/2603.15646}, note = {Source identifier: 2603.15646} }