@misc{indiciae711a0c54cb1a, title = {A Study on Knowledge Distillation from Weak Teacher for Scaling Up Pre-trained Language Models}, author = {Hayeon Lee and Rui Hou and Jongpil Kim and Davis Liang and Sung Ju Hwang and Alexander Min}, year = {2023}, url = {https://arxiv.org/abs/2305.18239}, note = {Source identifier: 2305.18239} }