@misc{indiciaec8c41fe3feb6, title = {Enhancing Multimodal Understanding with CLIP-Based Image-to-Text Transformation}, author = {Chang Che and Qunwei Lin and Xinyu Zhao and Jiaxin Huang and Liqiang Yu}, year = {2024}, url = {https://arxiv.org/abs/2401.06167}, note = {Source identifier: 2401.06167} }