@misc{indiciae97d52b935786, title = {HAVT-IVD: Heterogeneity-Aware Cross-Modal Network for Audio-Visual Surveillance: Idling Vehicles Detection With Multichannel Audio and Multiscale Visual Cues}, author = {Xiwen Li and Xiaoya Tang and Tolga Tasdizen}, year = {2025}, url = {https://arxiv.org/abs/2504.16102}, note = {Source identifier: 2504.16102} }