@misc{indiciaee6b6718bc00e, title = {Policy Split: Incentivizing Dual-Mode Exploration in LLM Reinforcement with Dual-Mode Entropy Regularization}, author = {Jiashu Yao and Heyan Huang and Daiqing Wu and Zeming Liu and Yuhang Guo}, year = {2026}, url = {https://arxiv.org/abs/2604.11510}, note = {Source identifier: 2604.11510} }