@article{compromisinghonestyandharmlessnessin, title = {Compromising Honesty and Harmlessness in Language Models via Deception Attacks}, author = {Laurène Vaugrante and Francesca Carlon and Maluna Menke and Thilo Hagendorff}, year = {2025}, eprint = {2502.08301}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2502.08301v2}, }