@article{doegocentricvideolanguagemodelscapturebo, title = {Do Egocentric Video-Language Models Capture Both Hand- and Object-Centric Cues?}, author = {Masatoshi Tateno and Alexandros Stergiou and Risa Shinoda and Yoichi Sato and Dima Damen}, year = {2026}, eprint = {2607.08514}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2607.08514}, }