@misc{indiciaea474e86337b6, title = {Rewarding What Matters: Step-by-Step Reinforcement Learning for Task-Oriented Dialogue}, author = {Huifang Du and Shuqin Li and Minghao Wu and Xuejing Feng and Yuan-Fang Li and Haofen Wang}, year = {2024}, url = {https://arxiv.org/abs/2406.14457}, note = {Source identifier: 2406.14457} }