@misc{indiciae1fe0bf795138, title = {On Convergence of Average-Reward Q-Learning in Weakly Communicating Markov Decision Processes}, author = {Yi Wan and Huizhen Yu and Richard S. Sutton}, year = {2024}, url = {https://arxiv.org/abs/2408.16262}, note = {Source identifier: 2408.16262} }