@article{marmotadeeplearningframeworkfor, title = {MARMOT: A Deep Learning Framework for Constructing Multimodal Representations for Vision-and-Language Tasks}, author = {Patrick Y. Wu and Walter R. Mebane Jr}, year = {2021}, eprint = {2109.11526}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2109.11526v1}, }