@misc{indiciaea7328c234ad2, title = {Shaking Up VLMs: Comparing Transformers and Structured State Space Models for Vision \& Language Modeling}, author = {Georgios Pantazopoulos and Malvina Nikandrou and Alessandro Suglia and Oliver Lemon and Arash Eshghi}, year = {2024}, url = {https://arxiv.org/abs/2409.05395}, note = {Source identifier: 2409.05395} }