@misc{indiciae311cc8e986e4, title = {Sparse Autoencoders Find Highly Interpretable Features in Language Models}, author = {Hoagy Cunningham and Aidan Ewart and Logan Riggs and Robert Huben and Lee Sharkey}, year = {2023}, url = {https://arxiv.org/abs/2309.08600}, note = {Source identifier: 2309.08600} }