@article{ontheexploitabilityofreinforcement, title = {RLHFPoison: Reward Poisoning Attack for Reinforcement Learning with Human Feedback in Large Language Models}, author = {Jiongxiao Wang and Junlin Wu and Muhao Chen and Yevgeniy Vorobeychik and Chaowei Xiao}, year = {2023}, eprint = {2311.09641}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2311.09641v2}, }