@misc{indiciaeb6e353cc6696, title = {Leveraging Foundation models for Unsupervised Audio-Visual Segmentation}, author = {Swapnil Bhosale and Haosen Yang and Diptesh Kanojia and Xiatian Zhu}, year = {2023}, url = {https://arxiv.org/abs/2309.06728}, note = {Source identifier: 2309.06728} }