@misc{indiciae3fa90f0c705f, title = {Enrich and Detect: Video Temporal Grounding with Multimodal LLMs}, author = {Shraman Pramanick and Effrosyni Mavroudi and Yale Song and Rama Chellappa and Lorenzo Torresani and Triantafyllos Afouras}, year = {2025}, url = {https://arxiv.org/abs/2510.17023}, note = {Source identifier: 2510.17023} }