@misc{indiciae72d0d2f58eb4, title = {SpeechCLIP+: Self-supervised multi-task representation learning for speech via CLIP and speech-image data}, author = {Hsuan-Fu Wang and Yi-Jen Shih and Heng-Jui Chang and Layne Berry and Puyuan Peng and Hung-yi Lee and Hsin-Min Wang and David Harwath}, year = {2024}, url = {https://arxiv.org/abs/2402.06959}, note = {Source identifier: 2402.06959} }