@inproceedings{rewardlearningfromhumanpreferencesand, title = {Reward learning from human preferences and demonstrations in Atari}, author = {Borja Ibarz and Jan Leike and Tobias Pohlen and Geoffrey Irving and Shane Legg and Dario Amodei}, year = {2018}, booktitle = {NeurIPS 2018 12}, eprint = {1811.06521}, archivePrefix = {arXiv}, url = {http://arxiv.org/abs/1811.06521v1}, }