@misc{indiciaebacc9fbdcf87, title = {How do Multimodal Foundation Models Encode Text and Speech? An Analysis of Cross-Lingual and Cross-Modal Representations}, author = {Hyunji Lee and Danni Liu and Supriti Sinhamahapatra and Jan Niehues}, year = {2025}, url = {https://arxiv.org/abs/2411.17666}, note = {Source identifier: 2411.17666} }