@misc{indiciae74c108673aaf, title = {AI Sandbagging: Language Models can Strategically Underperform on Evaluations}, author = {Teun van der Weij and Felix Hofstätter and Ollie Jaffe and Samuel F. Brown and Francis Rhys Ward}, year = {2025}, url = {https://arxiv.org/abs/2406.07358}, note = {Source identifier: 2406.07358} }