@misc{indiciae4f98ba0fd91e, title = {Model Tampering Attacks Enable More Rigorous Evaluations of LLM Capabilities}, author = {Zora Che and Stephen Casper and Robert Kirk and Anirudh Satheesh and Stewart Slocum and Lev E McKinney and Rohit Gandikota and Aidan Ewart and Domenic Rosati and Zichu Wu and Zikui Cai and Bilal Chughtai and Yarin Gal and Furong Huang and Dylan Hadfield-Menell}, year = {2025}, url = {https://arxiv.org/abs/2502.05209}, note = {Source identifier: 2502.05209} }