@misc{indiciae390b2e318a28, title = {Revisiting Audio-Visual Segmentation with Vision-Centric Transformer}, author = {Shaofei Huang and Rui Ling and Tianrui Hui and Hongyu Li and Xu Zhou and Shifeng Zhang and Si Liu and Richang Hong and Meng Wang}, year = {2025}, url = {https://arxiv.org/abs/2506.23623}, note = {Source identifier: 2506.23623} }