@misc{indiciaea953fbb22ff4, title = {Learning and Planning in Average-Reward Markov Decision Processes}, author = {Yi Wan and Abhishek Naik and Richard S. Sutton}, year = {2021}, url = {https://arxiv.org/abs/2006.16318}, note = {Source identifier: 2006.16318} }