@inproceedings{stvgbertavisuallinguistictransformer, title = {STVGBert: A Visual-Linguistic Transformer Based Framework for Spatio-Temporal Video Grounding}, author = {Rui Su and Qian Yu and Dong Xu}, year = {2021}, booktitle = {ICCV 2021 10}, url = {http://openaccess.thecvf.com//content/ICCV2021/html/Su_STVGBert_A_Visual-Linguistic_Transformer_Based_Framework_for_Spatio-Temporal_Video_Grounding_ICCV_2021_paper.html}, }