@misc{indiciae500a9ab92031, title = {Chain-of-Visual-Thought: Teaching VLMs to See and Think Better with Continuous Visual Tokens}, author = {Yiming Qin and Bomin Wei and Jiaxin Ge and Konstantinos Kallidromitis and Stephanie Fu and Trevor Darrell and XuDong Wang}, year = {2026}, url = {https://arxiv.org/abs/2511.19418}, note = {Source identifier: 2511.19418} }