@article{videomobileformervideorecognitionwith, title = {Video Mobile-Former: Video Recognition with Efficient Global Spatial-temporal Modeling}, author = {Rui Wang and Zuxuan Wu and Dongdong Chen and Yinpeng Chen and Xiyang Dai and Mengchen Liu and Luowei Zhou and Lu Yuan and Yu-Gang Jiang}, year = {2022}, eprint = {2208.12257}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2208.12257v1}, }