@misc{indiciae115416a6a9d1, title = {Rethinking Training-Inference Mismatch in LLM Reinforcement Learning: Where It Arises and How to Correct It}, author = {Tianrun Yu and Kaixiang Zhao and Shangzhe Li and Yuxiao Yang and Porter Jenkins and Weitong Zhang and Taylor W. Killian}, year = {2026}, url = {https://arxiv.org/abs/2609.32444}, note = {Source identifier: 2609.32444} }