@misc{indiciaeb211b4b2d046, title = {FM-IRL: Flow-Matching for Reward Modeling and Policy Regularization in Reinforcement Learning}, author = {Zhenglin Wan and Jingxuan Wu and Xingrui Yu and Chubin Zhang and Mingcong Lei and Bo An and Ivor Tsang}, year = {2026}, url = {https://arxiv.org/abs/2510.09222}, note = {Source identifier: 2510.09222} }