@misc{indiciae219f038a25e8, title = {DiFFPO: Training Diffusion LLMs to Reason Fast and Furious via Reinforcement Learning}, author = {Hanyang Zhao and Dawen Liang and Wenpin Tang and David Yao and Nathan Kallus}, year = {2026}, url = {https://arxiv.org/abs/2510.02212}, note = {Source identifier: 2510.02212} }