@misc{indiciaef804ed6ba266, title = {QCaption: Video Captioning and Q\&A through Fusion of Large Multimodal Models}, author = {Jiale Wang and Gee Wah Ng and Lee Onn Mak and Randall Cher and Ng Ding Hei Ryan and Davis Wang}, year = {2026}, doi = {10.23919/fusion59988.2024.10706514}, url = {https://arxiv.org/abs/2601.06566}, note = {Source identifier: 2601.06566} }