@misc{indiciae889673cf7ef1, title = {VicTR: Video-conditioned Text Representations for Activity Recognition}, author = {Kumara Kahatapitiya and Anurag Arnab and Arsha Nagrani and Michael S. Ryoo}, year = {2024}, url = {https://arxiv.org/abs/2304.02560}, note = {Source identifier: 2304.02560} }