@misc{indiciaea57d0f5bd340, title = {Train Large, Then Compress: Rethinking Model Size for Efficient Training and Inference of Transformers}, author = {Zhuohan Li and Eric Wallace and Sheng Shen and Kevin Lin and Kurt Keutzer and Dan Klein and Joseph E. Gonzalez}, year = {2020}, url = {https://arxiv.org/abs/2002.11794}, note = {Source identifier: 2002.11794} }