@misc{indiciae8e32c6841ee7, title = {Scaling the Long Video Understanding of Multimodal Large Language Models via Visual Memory Mechanism}, author = {Tao Chen and Kun Zhang and Qiong Wu and Xiao Chen and Chao Chang and Xiaoshuai Sun and Yiyi Zhou and Rongrong Ji}, year = {2026}, url = {https://arxiv.org/abs/2603.29252}, note = {Source identifier: 2603.29252} }