@misc{indiciae3f54b8734b09, title = {Cross modal video representations for weakly supervised active speaker localization}, author = {Rahul Sharma and Krishna Somandepalli and Shrikanth Narayanan}, year = {2021}, doi = {10.1109/tmm.2022.3229975}, url = {https://arxiv.org/abs/2003.04358}, note = {Source identifier: 2003.04358} }