@misc{indiciae69cd57ec802c, title = {Language-Guided Contrastive Audio-Visual Masked Autoencoder with Automatically Generated Audio-Visual-Text Triplets from Videos}, author = {Yuchi Ishikawa and Shota Nakada and Hokuto Munakata and Kazuhiro Saito and Tatsuya Komatsu and Yoshimitsu Aoki}, year = {2025}, url = {https://arxiv.org/abs/2507.11967}, note = {Source identifier: 2507.11967} }