@misc{indiciae34bb237ed136, title = {Audio-visual scene classification via contrastive event-object alignment and semantic-based fusion}, author = {Yuanbo Hou and Bo Kang and Dick Botteldooren}, year = {2022}, url = {https://arxiv.org/abs/2208.02086}, note = {Source identifier: 2208.02086} }