@misc{indiciaeff458735f22e, title = {Interpreting Learned Feedback Patterns in Large Language Models}, author = {Luke Marks and Amir Abdullah and Clement Neo and Rauno Arike and David Krueger and Philip Torr and Fazl Barez}, year = {2025}, url = {https://arxiv.org/abs/2310.08164}, note = {Source identifier: 2310.08164} }