@misc{indiciaee06788775cb8, title = {Reward Shaping with Recurrent Neural Networks for Speeding up On-Line Policy Learning in Spoken Dialogue Systems}, author = {Pei-Hao Su and David Vandyke and Milica Gasic and Nikola Mrksic and Tsung-Hsien Wen and Steve Young}, year = {2015}, url = {https://arxiv.org/abs/1508.03391}, note = {Source identifier: 1508.03391} }