@inproceedings{onlinetargetqlearningwithreverse1, title = {Online Target Q-learning with Reverse Experience Replay: Efficiently finding the Optimal Policy for Linear MDPs}, author = {Naman Agarwal and Syomantak Chaudhuri and Prateek Jain and Dheeraj Nagaraj and Praneeth Netrapalli}, year = {2021}, booktitle = {ICLR 2022 4}, eprint = {2110.08440}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2110.08440v2}, }