@misc{indiciae42798c9ffded, title = {On-line Active Reward Learning for Policy Optimisation in Spoken Dialogue Systems}, author = {Pei-Hao Su and Milica Gasic and Nikola Mrksic and Lina Rojas-Barahona and Stefan Ultes and David Vandyke and Tsung-Hsien Wen and Steve Young}, year = {2016}, url = {https://arxiv.org/abs/1605.07669}, note = {Source identifier: 1605.07669} }