@misc{indiciae8e9c9f60bc6a, title = {Stop Rewarding Hallucinated Steps: Faithfulness-Aware Step-Level Reinforcement Learning for Small Reasoning Models}, author = {Shuo Nie and Hexuan Deng and Chao Wang and Ruiyu Fang and Xuebo Liu and Shuangyong Song and Yu Li and Min Zhang and Xuelong Li}, year = {2026}, url = {https://arxiv.org/abs/2602.05897}, note = {Source identifier: 2602.05897} }