@misc{indiciaec2c850094ee4, title = {Exploring Pre-trained Text-to-Video Diffusion Models for Referring Video Object Segmentation}, author = {Zixin Zhu and Xuelu Feng and Dongdong Chen and Junsong Yuan and Chunming Qiao and Gang Hua}, year = {2024}, url = {https://arxiv.org/abs/2403.12042}, note = {Source identifier: 2403.12042} }