@misc{indiciaea6b765d02c06, title = {Online Target Q-learning with Reverse Experience Replay: Efficiently finding the Optimal Policy for Linear MDPs}, author = {Naman Agarwal and Syomantak Chaudhuri and Prateek Jain and Dheeraj Nagaraj and Praneeth Netrapalli}, year = {2021}, url = {https://arxiv.org/abs/2110.08440}, note = {Source identifier: 2110.08440} }