@misc{indiciaeb95343405b77, title = {Multi-CLIP: Contrastive Vision-Language Pre-training for Question Answering tasks in 3D Scenes}, author = {Alexandros Delitzas and Maria Parelli and Nikolas Hars and Georgios Vlassis and Sotirios Anagnostidis and Gregor Bachmann and Thomas Hofmann}, year = {2023}, url = {https://arxiv.org/abs/2306.02329}, note = {Source identifier: 2306.02329} }