@misc{indiciae66a53802e826, title = {Performative Policy Gradient: Optimality in Performative Reinforcement Learning}, author = {Debabrota Basu and Udvas Das and Brahim Driss and Uddalak Mukherjee}, year = {2026}, url = {https://arxiv.org/abs/2512.20576}, note = {Source identifier: 2512.20576} }