@misc{indiciaee9b0426d049a, title = {CAST: Modeling Visual State Transitions for Consistent Video Retrieval}, author = {Yanqing Liu and Yingcheng Liu and Fanghong Dong and Budianto Budianto and Cihang Xie and Yan Jiao}, year = {2026}, url = {https://arxiv.org/abs/2603.08648}, note = {Source identifier: 2603.08648} }