@misc{indiciaedd82f9a503da, title = {Large Language Models Generate Harmful Responses Using a Distinct Mechanism, Shared Across Harm Types}, author = {Hadas Orgad and Boyi Wei and Kaden Zheng and Martin Wattenberg and Peter Henderson and Seraphina Goldfarb-Tarrant and Yonatan Belinkov}, year = {2026}, url = {https://arxiv.org/abs/2604.09544}, note = {Source identifier: 2604.09544} }