@misc{indiciaeb645b1bbd9a1, title = {An Approach to Enriching Surgical Video Datasets for Fine-Grained Spatial-Temporal Understanding of Vision-Language Models}, author = {Lennart Maack and Alexander Schlaefer}, year = {2026}, url = {https://arxiv.org/abs/2604.00784}, note = {Source identifier: 2604.00784} }