@misc{indiciaea8f2a614f1a1, title = {Looking Similar, Sounding Different: Leveraging Counterfactual Cross-Modal Pairs for Audiovisual Representation Learning}, author = {Nikhil Singh and Chih-Wei Wu and Iroro Orife and Mahdi Kalayeh}, year = {2024}, url = {https://arxiv.org/abs/2304.05600}, note = {Source identifier: 2304.05600} }