@misc{indiciaee1343b0d0a19, title = {MaViLS, a Benchmark Dataset for Video-to-Slide Alignment, Assessing Baseline Accuracy with a Multimodal Alignment Algorithm Leveraging Speech, OCR, and Visual Features}, author = {Katharina Anderer and Andreas Reich and Matthias Wölfel}, year = {2024}, doi = {10.21437/interspeech.2024-978}, url = {https://arxiv.org/abs/2409.16765}, note = {Source identifier: 2409.16765} }