@misc{indiciae01f944b20d68, title = {Interpretable LLM Guardrails via Sparse Representation Steering}, author = {Zeqing He and Zhibo Wang and Huiyu Xu and Hejun Lin and Wenhui Zhang and Zhixuan Chu}, year = {2025}, url = {https://arxiv.org/abs/2503.16851}, note = {Source identifier: 2503.16851} }