@misc{indiciae4f1d566bb28a, title = {Align before Fuse: Vision and Language Representation Learning with Momentum Distillation}, author = {Junnan Li and Ramprasaath R. Selvaraju and Akhilesh Deepak Gotmare and Shafiq Joty and Caiming Xiong and Steven Hoi}, year = {2021}, url = {https://arxiv.org/abs/2107.07651}, note = {Source identifier: 2107.07651} }