@misc{indiciae7366c80be5f5, title = {Crab\$\textasciicircum{}\{+\}\$: A Scalable and Unified Audio-Visual Scene Understanding Model with Explicit Cooperation}, author = {Dongnuan Cai and Henghui Du and Chang Zhou and Xi Chen and Dan Guo and Hongyuan Zhang and Xuelong Li and Di Hu}, year = {2026}, url = {https://arxiv.org/abs/2603.04128}, note = {Source identifier: 2603.04128} }