@article{learningoptimaladvantagefrompreferences, title = {Learning Optimal Advantage from Preferences and Mistaking it for Reward}, author = {W. Bradley Knox and Stephane Hatgis-Kessell and Sigurdur Orn Adalgeirsson and Serena Booth and Anca Dragan and Peter Stone and Scott Niekum}, year = {2023}, eprint = {2310.02456}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2310.02456v1}, }