@misc{indiciae76ef01c3a6ce, title = {Unified-IO 2: Scaling Autoregressive Multimodal Models with Vision, Language, Audio, and Action}, author = {Jiasen Lu and Christopher Clark and Sangho Lee and Zichen Zhang and Savya Khosla and Ryan Marten and Derek Hoiem and Aniruddha Kembhavi}, year = {2023}, url = {https://arxiv.org/abs/2312.17172}, note = {Source identifier: 2312.17172} }