@article{leveraginggazeandsetofmarkinvllmsforhuma, title = {Leveraging Gaze and Set-of-Mark in VLLMs for Human-Object Interaction Anticipation from Egocentric Videos}, author = {Daniele Materia and Francesco Ragusa and Giovanni Maria Farinella}, year = {2026}, eprint = {2604.03667}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2604.03667}, }