@misc{indiciaebba5cbe1e961, title = {Multi-modal Large Language Model Enhanced Pseudo 3D Perception Framework for Visual Commonsense Reasoning}, author = {Jian Zhu and Hanli Wang and Miaojing Shi}, year = {2023}, url = {https://arxiv.org/abs/2301.13335}, note = {Source identifier: 2301.13335} }