@misc{indiciaef035192692b6, title = {Variational Policy Gradient Method for Reinforcement Learning with General Utilities}, author = {Junyu Zhang and Alec Koppel and Amrit Singh Bedi and Csaba Szepesvari and Mengdi Wang}, year = {2020}, url = {https://arxiv.org/abs/2007.02151}, note = {Source identifier: 2007.02151} }