@misc{indiciae5ff6f071d6e8, title = {Multimodal Abstractive Summarization of Instructional Videos with Vision-Language Models}, author = {Maham Nazir and Muhammad Aqeel and Richong Zhang and Francesco Setti}, year = {2026}, url = {https://arxiv.org/abs/2605.11959}, note = {Source identifier: 2605.11959} }