@article{diveintothescenebreakingtheperceptualbot, title = {Dive into the Scene: Breaking the Perceptual Bottleneck in Vision-Language Decision Making via Focus Plan Generation}, author = {Boyuan Xiao and Bohong Chen and Yumeng Li and Ji Feng and Yao-Xiang Ding and Kun Zhou}, year = {2026}, eprint = {2606.04046}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2606.04046}, }