@article{humanintheloopprovablyefficient, title = {Human-in-the-loop: Provably Efficient Preference-based Reinforcement Learning with General Function Approximation}, author = {Xiaoyu Chen and Han Zhong and Zhuoran Yang and Zhaoran Wang and LiWei Wang}, year = {2022}, eprint = {2205.11140}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2205.11140v2}, }