@misc{indiciaef9a0b7d653dd, title = {Video Active Perception: Effective Inference-Time Long-Form Video Understanding with Vision-Language Models}, author = {Martin Q. Ma and Willis Guo and Aditya Agrawal and Ankit Gupta and Paul Pu Liang and Ruslan Salakhutdinov and Louis-Philippe Morency}, year = {2026}, url = {https://arxiv.org/abs/2605.01662}, note = {Source identifier: 2605.01662} }