@misc{indiciae7daee7b8900e, title = {Interpretable Steering of Large Language Models with Feature Guided Activation Additions}, author = {Samuel Soo and Chen Guang and Wesley Teng and Chandrasekaran Balaganesh and Tan Guoxian and Yan Ming}, year = {2025}, url = {https://arxiv.org/abs/2501.09929}, note = {Source identifier: 2501.09929} }