@misc{indiciae7af10a8e6acb, title = {Vid2Seq: Large-Scale Pretraining of a Visual Language Model for Dense Video Captioning}, author = {Antoine Yang and Arsha Nagrani and Paul Hongsuck Seo and Antoine Miech and Jordi Pont-Tuset and Ivan Laptev and Josef Sivic and Cordelia Schmid}, year = {2023}, url = {https://arxiv.org/abs/2302.14115}, note = {Source identifier: 2302.14115} }