@misc{indiciaef0cfc515aa3e, title = {Learning from an Exploring Demonstrator: Optimal Reward Estimation for Bandits}, author = {Wenshuo Guo and Kumar Krishna Agrawal and Aditya Grover and Vidya Muthukumar and Ashwin Pananjady}, year = {2022}, url = {https://arxiv.org/abs/2106.14866}, note = {Source identifier: 2106.14866} }