@misc{indiciae7bc947cc151e, title = {Deep Music Retrieval for Fine-Grained Videos by Exploiting Cross-Modal-Encoded Voice-Overs}, author = {Tingtian Li and Zixun Sun and Haoruo Zhang and Jin Li and Ziming Wu and Hui Zhan and Yipeng Yu and Hengcan Shi}, year = {2021}, doi = {10.1145/3404835.3462993}, url = {https://arxiv.org/abs/2104.10557}, note = {Source identifier: 2104.10557} }