@misc{indiciae639f8bf034f4, title = {MIST: Multi-modal Iterative Spatial-Temporal Transformer for Long-form Video Question Answering}, author = {Difei Gao and Luowei Zhou and Lei Ji and Linchao Zhu and Yi Yang and Mike Zheng Shou}, year = {2022}, url = {https://arxiv.org/abs/2212.09522}, note = {Source identifier: 2212.09522} }