@misc{indiciaeace74559b99a, title = {Transformers Don't Need LayerNorm at Inference Time: Scaling LayerNorm Removal to GPT-2 XL and the Implications for Mechanistic Interpretability}, author = {Luca Baroni and Galvin Khara and Joachim Schaeffer and Marat Subkhankulov and Stefan Heimersheim}, year = {2025}, url = {https://arxiv.org/abs/2507.02559}, note = {Source identifier: 2507.02559} }