@misc{indiciae7e8dcea5bc40, title = {Convergence of Finite Memory Q-Learning for POMDPs and Near Optimality of Learned Policies under Filter Stability}, author = {Ali Devran Kara and Serdar Yuksel}, year = {2022}, url = {https://arxiv.org/abs/2103.12158}, note = {Source identifier: 2103.12158} }