@article{offpolicyreinforcementlearningwith2, title = {Off-policy Reinforcement Learning with Optimistic Exploration and Distribution Correction}, author = {Jiachen Li and Shuo Cheng and Zhenyu Liao and Huayan Wang and William Yang Wang and Qinxun Bai}, year = {2021}, eprint = {2110.12081}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2110.12081v3}, }