@misc{indiciae8b63bdd0ffb8, title = {Representation Learning for Semantic Alignment of Language, Audio, and Visual Modalities}, author = {Parthasaarathy Sudarsanam and Irene Martín-Morató and Tuomas Virtanen}, year = {2025}, url = {https://arxiv.org/abs/2505.14562}, note = {Source identifier: 2505.14562} }