@misc{indiciae92913e51d642, title = {SpeechCT-CLIP: Distilling Text-Image Knowledge to Speech for Voice-Native Multimodal CT Analysis}, author = {Lukas Buess and Jan Geier and David Bani-Harouni and Chantal Pellegrini and Matthias Keicher and Paula Andrea Perez-Toro and Nassir Navab and Andreas Maier and Tomas Arias-Vergara}, year = {2025}, url = {https://arxiv.org/abs/2510.02322}, note = {Source identifier: 2510.02322} }