@misc{indiciae49000fa5055d, title = {Shotluck Holmes: A Family of Efficient Small-Scale Large Language Vision Models For Video Captioning and Summarization}, author = {Richard Luo and Austin Peng and Adithya Vasudev and Rishabh Jain}, year = {2024}, doi = {10.1145/3689091.3690086}, url = {https://arxiv.org/abs/2405.20648}, note = {Source identifier: 2405.20648} }