@misc{indiciaee9f407f9957f, title = {IPFormer-VideoLLM: Enhancing Multi-modal Video Understanding for Multi-shot Scenes}, author = {Yujia Liang and Jile Jiao and Xuetao Feng and Zixuan Ye and Yuan Wang and Zhicheng Wang}, year = {2025}, url = {https://arxiv.org/abs/2506.21116}, note = {Source identifier: 2506.21116} }