@article{whatdovisionlanguagemodelsseeinthecontex, title = {What do vision-language models see in the context? Investigating multimodal in-context learning}, author = {Gabriel O. dos Santos and Esther Colombini and Sandra Avila}, year = {2025}, eprint = {2510.24331}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2510.24331}, }