@misc{indiciae6568815846c5, title = {Batch Policy Learning in Average Reward Markov Decision Processes}, author = {Peng Liao and Zhengling Qi and Runzhe Wan and Predrag Klasnja and Susan Murphy}, year = {2022}, url = {https://arxiv.org/abs/2007.11771}, note = {Source identifier: 2007.11771} }