@misc{indiciae1db29ddd0403, title = {Expected Return Causes Outcome-Level Mode Collapse in Reinforcement Learning and How to Fix It with Inverse Probability Scaling}, author = {Abhijeet Sinha and Sundari Elango and Dianbo Liu}, year = {2026}, url = {https://arxiv.org/abs/2601.21669}, note = {Source identifier: 2601.21669} }