@misc{indiciae12a5fe1af39e, title = {Rethinking Cross-modal Interaction from a Top-down Perspective for Referring Video Object Segmentation}, author = {Chen Liang and Yu Wu and Tianfei Zhou and Wenguan Wang and Zongxin Yang and Yunchao Wei and Yi Yang}, year = {2024}, url = {https://arxiv.org/abs/2106.01061}, note = {Source identifier: 2106.01061} }