@misc{indiciaeaa3fff12e2e0, title = {Making Bias Non-Predictive: Training Robust LLM Reasoning via Reinforcement Learning}, author = {Qian Wang and Xuandong Zhao and Zirui Zhang and Zhanzhi Lou and Nuo Chen and Dawn Song and Bingsheng He}, year = {2026}, url = {https://arxiv.org/abs/2602.01528}, note = {Source identifier: 2602.01528} }