@article{tradeoffsbetweenalignmentandhelpfulness, title = {Tradeoffs Between Alignment and Helpfulness in Language Models with Representation Engineering}, author = {Yotam Wolf and Noam Wies and Dorin Shteyman and Binyamin Rothberg and Yoav Levine and Amnon Shashua}, year = {2024}, eprint = {2401.16332}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2401.16332v4}, }