@misc{indiciae6c27e6cfda5f, title = {The Distributional Hypothesis Does Not Fully Explain the Benefits of Masked Language Model Pretraining}, author = {Ting-Rui Chiang and Dani Yogatama}, year = {2023}, url = {https://arxiv.org/abs/2310.16261}, note = {Source identifier: 2310.16261} }