@misc{indiciae0aed889c12eb, title = {ViLLa: Video Reasoning Segmentation with Large Language Model}, author = {Rongkun Zheng and Lu Qi and Xi Chen and Yi Wang and Kun Wang and Yu Qiao and Hengshuang Zhao}, year = {2025}, url = {https://arxiv.org/abs/2407.14500}, note = {Source identifier: 2407.14500} }