@article{deliberativealignmentisdeepbutuncertaint, title = {Deliberative Alignment is Deep, but Uncertainty Remains: Inference time safety improvement in reasoning via attribution of unsafe behavior to base model}, author = {Pankayaraj Pathmanathan and Furong Huang}, year = {2026}, eprint = {2604.09665}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2604.09665}, }