@misc{indiciae1f26bbdba508, title = {Can Local Vision-Language Models improve Activity Recognition over Vision Transformers? -- Case Study on Newborn Resuscitation}, author = {Enrico Guerriero and Kjersti Engan and Øyvind Meinich-Bache}, year = {2026}, url = {https://arxiv.org/abs/2602.12002}, note = {Source identifier: 2602.12002} }