@misc{indiciaee7f1e2fd75b7, title = {V2Xum-LLM: Cross-Modal Video Summarization with Temporal Prompt Instruction Tuning}, author = {Hang Hua and Yolo Yunlong Tang and Chenliang Xu and Jiebo Luo}, year = {2025}, url = {https://arxiv.org/abs/2404.12353}, note = {Source identifier: 2404.12353} }