@misc{indiciae6640d70ff5a1, title = {LLaVA-CoT: Let Vision Language Models Reason Step-by-Step}, author = {Guowei Xu and Peng Jin and Ziang Wu and Hao Li and Yibing Song and Lichao Sun and Li Yuan}, year = {2025}, url = {https://arxiv.org/abs/2411.10440}, note = {Source identifier: 2411.10440} }