@misc{indiciae887889edf13f, title = {The Visual Bottleneck: Sparse-Frame Adaptation of MLLMs for Joint Spatial-Temporal Video Grounding}, author = {Jiameng Zhang and Srikanth Madikeri}, year = {2026}, url = {https://arxiv.org/abs/2607.24570}, note = {Source identifier: 2607.24570} }