@misc{indiciae4ddda990c1d4, title = {Beyond Generic: Enhancing Image Captioning with Real-World Knowledge using Vision-Language Pre-Training Model}, author = {Kanzhi Cheng and Wenpo Song and Zheng Ma and Wenhao Zhu and Zixuan Zhu and Jianbing Zhang}, year = {2023}, url = {https://arxiv.org/abs/2308.01126}, note = {Source identifier: 2308.01126} }