@misc{indiciae7fd8d3f290f0, title = {DynImg: Key Frames with Visual Prompts are Good Representation for Multi-Modal Video Understanding}, author = {Xiaoyi Bao and Chenwei Xie and Hao Tang and Tingyu Weng and Xiaofeng Wang and Yun Zheng and Xingang Wang}, year = {2025}, url = {https://arxiv.org/abs/2507.15569}, note = {Source identifier: 2507.15569} }