@misc{indiciae844f1ca79e2b, title = {MARMOT: A Deep Learning Framework for Constructing Multimodal Representations for Vision-and-Language Tasks}, author = {Patrick Y. Wu and Walter R. Mebane Jr}, year = {2021}, url = {https://arxiv.org/abs/2109.11526}, note = {Source identifier: 2109.11526} }