@article{exploitingunlabeleddataforfeedback, title = {Exploiting Unlabeled Data for Feedback Efficient Human Preference based Reinforcement Learning}, author = {Mudit Verma and Siddhant Bhambri and Subbarao Kambhampati}, year = {2023}, eprint = {2302.08738}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2302.08738v1}, }