@misc{indiciae03f0d65ec6b4, title = {Towards Practical and Efficient Image-to-Speech Captioning with Vision-Language Pre-training and Multi-modal Tokens}, author = {Minsu Kim and Jeongsoo Choi and Soumi Maiti and Jeong Hun Yeo and Shinji Watanabe and Yong Man Ro}, year = {2023}, url = {https://arxiv.org/abs/2309.08531}, note = {Source identifier: 2309.08531} }