@misc{indiciaefa4e3739cfea, title = {MM-ViT: Multi-Modal Video Transformer for Compressed Video Action Recognition}, author = {Jiawei Chen and Chiu Man Ho}, year = {2021}, url = {https://arxiv.org/abs/2108.09322}, note = {Source identifier: 2108.09322} }