@misc{indiciaeb2890bfe5707, title = {Scalable and Accurate Self-supervised Multimodal Representation Learning without Aligned Video and Text Data}, author = {Vladislav Lialin and Stephen Rawls and David Chan and Shalini Ghosh and Anna Rumshisky and Wael Hamza}, year = {2023}, doi = {10.1109/wacvw58289.2023.00043}, url = {https://arxiv.org/abs/2304.02080}, note = {Source identifier: 2304.02080} }