@misc{indiciae3838602b1ede, title = {Cognitive models can reveal interpretable value trade-offs in language models}, author = {Sonia K. Murthy and Rosie Zhao and Jennifer Hu and Sham Kakade and Markus Wulfmeier and Peng Qian and Tomer Ullman}, year = {2026}, url = {https://arxiv.org/abs/2506.20666}, note = {Source identifier: 2506.20666} }