@article{nanonestedhumaninthelooprewardlearning, title = {Nano: Nested Human-in-the-Loop Reward Learning for Few-shot Language Model Control}, author = {Xiang Fan and Yiwei Lyu and Paul Pu Liang and Ruslan Salakhutdinov and Louis-Philippe Morency}, year = {2022}, eprint = {2211.05750}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2211.05750v3}, }