@misc{indiciaec0437eb14347, title = {Leveraging multimodal explanatory annotations for video interpretation with Modality Specific Dataset}, author = {Elisa Ancarani and Julie Tores and Lucile Sassatelli and Rémy Sun and Hui-Yin Wu and Frédéric Precioso}, year = {2025}, url = {https://arxiv.org/abs/2504.11232}, note = {Source identifier: 2504.11232} }