@article{reinforcementlearningfromdiversehuman, title = {Reinforcement Learning from Diverse Human Preferences}, author = {Wanqi Xue and Bo An and Shuicheng Yan and Zhongwen Xu}, year = {2023}, eprint = {2301.11774}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2301.11774v3}, }