@misc{indiciae7cf457bf5ce5, title = {R-4B: Incentivizing General-Purpose Auto-Thinking Capability in MLLMs via Bi-Mode Annealing and Reinforce Learning}, author = {Qi Yang and Bolin Ni and Shiming Xiang and Han Hu and Houwen Peng and Jie Jiang}, year = {2025}, url = {https://arxiv.org/abs/2508.21113}, note = {Source identifier: 2508.21113} }