If you use this dataset please cite us at
@misc{pathmanathan2026deliberativealignmentdeepuncertainty,
title={Deliberative Alignment is Deep, but Uncertainty Remains: Inference time safety improvement in reasoning via attribution of unsafe behavior to base model},
author={Pankayaraj Pathmanathan and Furong Huang},
year={2026},
eprint={2604.09665},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={
https://arxiv.org/abs/2604.09665},
}