@misc{indiciae106d9174d145, title = {ConsensusDrop: Fusing Visual and Cross-Modal Saliency for Efficient Vision Language Models}, author = {Dhruv Parikh and Haoyang Fan and Rajgopal Kannan and Viktor Prasanna}, year = {2026}, url = {https://arxiv.org/abs/2602.00946}, note = {Source identifier: 2602.00946} }