@misc{indiciaef7626e93e555, title = {What, when, and where? -- Self-Supervised Spatio-Temporal Grounding in Untrimmed Multi-Action Videos from Narrated Instructions}, author = {Brian Chen and Nina Shvetsova and Andrew Rouditchenko and Daniel Kondermann and Samuel Thomas and Shih-Fu Chang and Rogerio Feris and James Glass and Hilde Kuehne}, year = {2024}, url = {https://arxiv.org/abs/2303.16990}, note = {Source identifier: 2303.16990} }