@article{qxploreqlearningexplorationbymaximizing, title = {Reward Prediction Error as an Exploration Objective in Deep RL}, author = {Riley Simmons-Edler and Ben Eisner and Daniel Yang and Anthony Bisulco and Eric Mitchell and Sebastian Seung and Daniel Lee}, year = {2019}, eprint = {1906.08189}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/1906.08189v5}, }