@misc{indiciaed825e4f611c1, title = {End-to-End Spatio-Temporal Action Localisation with Video Transformers}, author = {Alexey Gritsenko and Xuehan Xiong and Josip Djolonga and Mostafa Dehghani and Chen Sun and Mario Lučić and Cordelia Schmid and Anurag Arnab}, year = {2023}, url = {https://arxiv.org/abs/2304.12160}, note = {Source identifier: 2304.12160} }