@misc{indiciaee826b9e8bbc9, title = {SViTT: Temporal Learning of Sparse Video-Text Transformers}, author = {Yi Li and Kyle Min and Subarna Tripathi and Nuno Vasconcelos}, year = {2023}, url = {https://arxiv.org/abs/2304.08809}, note = {Source identifier: 2304.08809} }