@article{honestytosubterfugeincontext, title = {Honesty to Subterfuge: In-Context Reinforcement Learning Can Make Honest Models Reward Hack}, author = {Leo McKee-Reid and Christoph Sträter and Maria Angelica Martinez and Joe Needham and Mikita Balesni}, year = {2024}, eprint = {2410.06491}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2410.06491v1}, }