@misc{indiciae81d939a09e2f, title = {Convergence of a Human-in-the-Loop Policy-Gradient Algorithm With Eligibility Trace Under Reward, Policy, and Advantage Feedback}, author = {Ishaan Shah and David Halpern and Kavosh Asadi and Michael L. Littman}, year = {2021}, url = {https://arxiv.org/abs/2109.07054}, note = {Source identifier: 2109.07054} }