@misc{indiciae5173f78d4562, title = {nmT5 -- Is parallel data still relevant for pre-training massively multilingual language models?}, author = {Mihir Kale and Aditya Siddhant and Noah Constant and Melvin Johnson and Rami Al-Rfou and Linting Xue}, year = {2021}, url = {https://arxiv.org/abs/2106.02171}, note = {Source identifier: 2106.02171} }