@misc{indiciae89e2373bb261, title = {OctoThinker: Mid-training Incentivizes Reinforcement Learning Scaling}, author = {Zengzhi Wang and Fan Zhou and Xuefeng Li and Pengfei Liu}, year = {2025}, url = {https://arxiv.org/abs/2506.20512}, note = {Source identifier: 2506.20512} }