@misc{indiciae6fa75169cb40, title = {When Vision Becomes Text: Visual Token Pruning via Cross-Modal Residual Guidance in VLMs}, author = {Congyang Ou and Ruike Song and Yang Zhou and Libo Sun and Haokui Zhang and Zhenbo Luo}, year = {2026}, url = {https://arxiv.org/abs/2608.10489}, note = {Source identifier: 2608.10489} }