@misc{indiciaedb4d7fcdd444, title = {SmolDocling: An ultra-compact vision-language model for end-to-end multi-modal document conversion}, author = {Ahmed Nassar and Andres Marafioti and Matteo Omenetti and Maksym Lysak and Nikolaos Livathinos and Christoph Auer and Lucas Morin and Rafael Teixeira de Lima and Yusik Kim and A. Said Gurbuz and Michele Dolfi and Miquel Farré and Peter W. J. Staar}, year = {2025}, url = {https://arxiv.org/abs/2503.11576}, note = {Source identifier: 2503.11576} }