@misc{indiciaeb480f7591e10, title = {Generalizing Trust: Weak-to-Strong Trustworthiness in Language Models}, author = {Martin Pawelczyk and Lillian Sun and Zhenting Qi and Aounon Kumar and Himabindu Lakkaraju}, year = {2024}, url = {https://arxiv.org/abs/2501.00418}, note = {Source identifier: 2501.00418} }