@misc{indiciaebd9b5b431d56, title = {On the Global Optimality of Policy Gradient Methods in General Utility Reinforcement Learning}, author = {Anas Barakat and Souradip Chakraborty and Peihong Yu and Pratap Tokekar and Amrit Singh Bedi}, year = {2025}, url = {https://arxiv.org/abs/2410.04108}, note = {Source identifier: 2410.04108} }