@misc{indiciae0ad6b63853b4, title = {Audio Visual Scene-Aware Dialog Generation with Transformer-based Video Representations}, author = {Yoshihiro Yamazaki and Shota Orihashi and Ryo Masumura and Mihiro Uchida and Akihiko Takashima}, year = {2022}, url = {https://arxiv.org/abs/2202.09979}, note = {Source identifier: 2202.09979} }