@misc{indiciae98b0e221d4a3, title = {Perception Tokens Enhance Visual Reasoning in Multimodal Language Models}, author = {Mahtab Bigverdi and Zelun Luo and Cheng-Yu Hsieh and Ethan Shen and Dongping Chen and Linda G. Shapiro and Ranjay Krishna}, year = {2024}, url = {https://arxiv.org/abs/2412.03548}, note = {Source identifier: 2412.03548} }