@misc{indiciae0ac03c88fabc, title = {AVFormer: Injecting Vision into Frozen Speech Models for Zero-Shot AV-ASR}, author = {Paul Hongsuck Seo and Arsha Nagrani and Cordelia Schmid}, year = {2023}, url = {https://arxiv.org/abs/2303.16501}, note = {Source identifier: 2303.16501} }