@misc{indiciae542e0c62b0ea, title = {HAVEN: Hierarchically Aligned Multimodal Benchmark for Unified Video Understanding}, author = {Mengqi Shi and Haopeng Zhang}, year = {2026}, url = {https://arxiv.org/abs/2605.19223}, note = {Source identifier: 2605.19223} }