@misc{indiciaef0e1b4ff2098, title = {ViCaS: A Dataset for Combining Holistic and Pixel-level Video Understanding using Captions with Grounded Segmentation}, author = {Ali Athar and Xueqing Deng and Liang-Chieh Chen}, year = {2025}, url = {https://arxiv.org/abs/2412.09754}, note = {Source identifier: 2412.09754} }