@misc{indiciae33d138041cdc, title = {Mechanistic Interpretability for AI Safety -- A Review}, author = {Leonard Bereska and Efstratios Gavves}, year = {2024}, url = {https://arxiv.org/abs/2404.14082}, note = {Source identifier: 2404.14082} }