@misc{indiciae30f2f35a5492, title = {Greedification Operators for Policy Optimization: Investigating Forward and Reverse KL Divergences}, author = {Alan Chan and Hugo Silva and Sungsu Lim and Tadashi Kozuno and A. Rupam Mahmood and Martha White}, year = {2022}, url = {https://arxiv.org/abs/2107.08285}, note = {Source identifier: 2107.08285} }