@misc{indiciaefa993a3e7786, title = {Learning Policy from a Single Trajectory in Average-Reward Markov Decision Process}, author = {Jongmin Lee and Ernest K. Ryu and Vaneet Aggarwal}, year = {2026}, url = {https://arxiv.org/abs/2606.16729}, note = {Source identifier: 2606.16729} }