@misc{indiciae6cf7d75c85c4, title = {Vision-Speech Models: Teaching Speech Models to Converse about Images}, author = {Amélie Royer and Moritz Böhle and Gabriel de Marmiesse and Laurent Mazaré and Neil Zeghidour and Alexandre Défossez and Patrick Pérez}, year = {2025}, url = {https://arxiv.org/abs/2503.15633}, note = {Source identifier: 2503.15633} }