@misc{indiciae018fd446d916, title = {Human-centric Spatio-Temporal Video Grounding With Visual Transformers}, author = {Zongheng Tang and Yue Liao and Si Liu and Guanbin Li and Xiaojie Jin and Hongxu Jiang and Qian Yu and Dong Xu}, year = {2021}, url = {https://arxiv.org/abs/2011.05049}, note = {Source identifier: 2011.05049} }