@misc{indiciae7b901acfb3a0, title = {Learning Spatial Features from Audio-Visual Correspondence in Egocentric Videos}, author = {Sagnik Majumder and Ziad Al-Halah and Kristen Grauman}, year = {2024}, url = {https://arxiv.org/abs/2307.04760}, note = {Source identifier: 2307.04760} }