@misc{indiciaefd1c063287aa, title = {Reinforcement Learning in Factored MDPs: Oracle-Efficient Algorithms and Tighter Regret Bounds for the Non-Episodic Setting}, author = {Ziping Xu and Ambuj Tewari}, year = {2020}, url = {https://arxiv.org/abs/2002.02302}, note = {Source identifier: 2002.02302} }