@misc{indiciae7b16718850cb, title = {See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding}, author = {Boyuan Sun and Bowen Yin and Yuanming Li and Xihan Wei and Qibin Hou}, year = {2026}, url = {https://arxiv.org/abs/2605.18018}, note = {Source identifier: 2605.18018} }