@misc{indiciae1afc40786959, title = {Reinforcement Learning Fine-tuning of Language Models is Biased Towards More Extractable Features}, author = {Diogo Cruz and Edoardo Pona and Alex Holness-Tofts and Elias Schmied and VĂ­ctor Abia Alonso and Charlie Griffin and Bogdan-Ionut Cirstea}, year = {2023}, url = {https://arxiv.org/abs/2311.04046}, note = {Source identifier: 2311.04046} }