@misc{indiciae1a6dda730f40, title = {Mosaic: Cross-Modal Clustering for Efficient Video Understanding}, author = {Tuowei Wang and He Zhou and Chengru Song and Qiushi Li and Ju Ren}, year = {2026}, url = {https://arxiv.org/abs/2604.10060}, note = {Source identifier: 2604.10060} }