@misc{indiciaecd668d1aa8d2, title = {Deliberative Alignment is Deep, but Uncertainty Remains: Inference time safety improvement in reasoning via attribution of unsafe behavior to base model}, author = {Pankayaraj Pathmanathan and Furong Huang}, year = {2026}, url = {https://arxiv.org/abs/2604.09665}, note = {Source identifier: 2604.09665} }