@misc{indiciae913e7e891aab, title = {Rethink Cross-Modal Fusion in Weakly-Supervised Audio-Visual Video Parsing}, author = {Yating Xu and Conghui Hu and Gim Hee Lee}, year = {2023}, url = {https://arxiv.org/abs/2311.08151}, note = {Source identifier: 2311.08151} }