@misc{indiciaeb5eaa30e8143, title = {ST-SimDiff: Balancing Spatiotemporal Similarity and Difference for Efficient Video Understanding with MLLMs}, author = {Bingjun Luo and Tony Wang and Chaoqi Chen and Xinpeng Ding}, year = {2026}, url = {https://arxiv.org/abs/2605.22158}, note = {Source identifier: 2605.22158} }