@misc{indiciae3988d581c63c, title = {Deep Learning Based Multi-modal Addressee Recognition in Visual Scenes with Utterances}, author = {Thao Minh Le and Nobuyuki Shimizu and Takashi Miyazaki and Koichi Shinoda}, year = {2018}, doi = {10.24963/ijcai.2018/214}, url = {https://arxiv.org/abs/1809.04288}, note = {Source identifier: 1809.04288} }