@misc{indiciae363b80f5b772, title = {PolyViT: Co-training Vision Transformers on Images, Videos and Audio}, author = {Valerii Likhosherstov and Anurag Arnab and Krzysztof Choromanski and Mario Lucic and Yi Tay and Adrian Weller and Mostafa Dehghani}, year = {2021}, url = {https://arxiv.org/abs/2111.12993}, note = {Source identifier: 2111.12993} }