@misc{indiciae67a372663eab, title = {GRAIL: Gradient-Reweighted Advantages for Reinforcement Learning with Verifiable Rewards}, author = {Tej Deep Pala and Vernon Toh and Soujanya Poria}, year = {2026}, url = {https://arxiv.org/abs/2606.04889}, note = {Source identifier: 2606.04889} }