@misc{indiciaea2e2702144ce, title = {Learning Reward Functions from Diverse Sources of Human Feedback: Optimally Integrating Demonstrations and Preferences}, author = {Erdem Bıyık and Dylan P. Losey and Malayandi Palan and Nicholas C. Landolfi and Gleb Shevchuk and Dorsa Sadigh}, year = {2021}, url = {https://arxiv.org/abs/2006.14091}, note = {Source identifier: 2006.14091} }