@misc{indiciae10ce9cffa47a, title = {Towers of Babel: Combining Images, Language, and 3D Geometry for Learning Multimodal Vision}, author = {Xiaoshi Wu and Hadar Averbuch-Elor and Jin Sun and Noah Snavely}, year = {2021}, url = {https://arxiv.org/abs/2108.05863}, note = {Source identifier: 2108.05863} }