@misc{indiciae520a80f6d5d6, title = {CMTM: Cross-Modal Token Modulation for Unsupervised Video Object Segmentation}, author = {Inseok Jeon and Suhwan Cho and Minhyeok Lee and Seunghoon Lee and Minseok Kang and Jungho Lee and Chaewon Park and Donghyeong Kim and Sangyoun Lee}, year = {2026}, url = {https://arxiv.org/abs/2604.14630}, note = {Source identifier: 2604.14630} }