@misc{indiciae36bb536d7b4b, title = {Distilling Internet-Scale Vision-Language Models into Embodied Agents}, author = {Theodore Sumers and Kenneth Marino and Arun Ahuja and Rob Fergus and Ishita Dasgupta}, year = {2023}, url = {https://arxiv.org/abs/2301.12507}, note = {Source identifier: 2301.12507} }