@misc{indiciae4516d09f7c43, title = {SynthVLM: Towards High-Quality and Efficient Synthesis of Image-Caption Datasets for Vision-Language Models}, author = {Zheng Liu and Hao Liang and Bozhou Li and Wentao Xiong and Chong Chen and Conghui He and Wentao Zhang and Bin Cui}, year = {2025}, url = {https://arxiv.org/abs/2407.20756}, note = {Source identifier: 2407.20756} }