@article{villavideoreasoningsegmentationwithlarge, title = {ViLLa: Video Reasoning Segmentation with Large Language Model}, author = {Rongkun Zheng and Lu Qi and Xi Chen and Yi Wang and Kun Wang and Yu Qiao and Hengshuang Zhao}, year = {2024}, eprint = {2407.14500}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2407.14500v2}, }