@misc{indiciae3e85ee5db7fc, title = {Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog}, author = {Natasha Jaques and Asma Ghandeharioun and Judy Hanwen Shen and Craig Ferguson and Agata Lapedriza and Noah Jones and Shixiang Gu and Rosalind Picard}, year = {2019}, url = {https://arxiv.org/abs/1907.00456}, note = {Source identifier: 1907.00456} }