@misc{indiciaed401bd96f0f9, title = {Where to Focus: Query-Modulated Multimodal Keyframe Selection for Long Video Understanding}, author = {Shaoguang Wang and Weiyu Guo and Ziyang Chen and Xuming Hu and Hui Xiong}, year = {2026}, doi = {10.1145/3767308.3835040}, url = {https://arxiv.org/abs/2604.17422}, note = {Source identifier: 2604.17422} }