@article{roverrecursivereasoningovervideoswithvis, title = {ROVER: Recursive Reasoning Over Videos with Vision-Language Models for Embodied Tasks}, author = {Philip Schroeder and Ondrej Biza and Thomas Weng and Hongyin Luo and James Glass}, year = {2025}, eprint = {2508.01943}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2508.01943}, }