@misc{indiciae345dfd23b67f, title = {Captions Speak Louder than Images: Generalizing Foundation Models for E-commerce from High-quality Multimodal Instruction Data}, author = {Xinyi Ling and Hanwen Du and Bo Peng and Zhihui Zhu and Xia Ning}, year = {2025}, url = {https://arxiv.org/abs/2410.17337}, note = {Source identifier: 2410.17337} }