@misc{indiciae6a097733015c, title = {Building and better understanding vision-language models: insights and future directions}, author = {Hugo Laurençon and Andrés Marafioti and Victor Sanh and Léo Tronchon}, year = {2024}, url = {https://arxiv.org/abs/2408.12637}, note = {Source identifier: 2408.12637} }