@misc{indiciae3eb977d21f03, title = {SpeechCLIP: Integrating Speech with Pre-Trained Vision and Language Model}, author = {Yi-Jen Shih and Hsuan-Fu Wang and Heng-Jui Chang and Layne Berry and Hung-yi Lee and David Harwath}, year = {2022}, url = {https://arxiv.org/abs/2210.00705}, note = {Source identifier: 2210.00705} }