@misc{indiciae761ba650d194, title = {DETECLAP: Enhancing Audio-Visual Representation Learning with Object Information}, author = {Shota Nakada and Taichi Nishimura and Hokuto Munakata and Masayoshi Kondo and Tatsuya Komatsu}, year = {2024}, url = {https://arxiv.org/abs/2409.11729}, note = {Source identifier: 2409.11729} }