@misc{indiciaee0c38bd0c0e6, title = {GroundVTS: Visual Token Sampling in Multimodal Large Language Models for Video Temporal Grounding}, author = {Rong Fan and Kaiyan Xiao and Minghao Zhu and Liuyi Wang and Kai Dai and Zhao Yang}, year = {2026}, url = {https://arxiv.org/abs/2604.02093}, note = {Source identifier: 2604.02093} }