@misc{indiciae7915d1d3bf14, title = {FlashVID: Efficient Video Large Language Models via Training-free Tree-based Spatiotemporal Token Merging}, author = {Ziyang Fan and Keyu Chen and Ruilong Xing and Yulin Li and Li Jiang and Zhuotao Tian}, year = {2026}, url = {https://arxiv.org/abs/2602.08024}, note = {Source identifier: 2602.08024} }