@article{llavaspenhancingvisualrepresentationwith, title = {LLaVA-SP: Enhancing Visual Representation with Visual Spatial Tokens for MLLMs}, author = {Haoran Lou and Chunxiao Fan and Ziyan Liu and Yuexin Wu and Xinxiang Wang}, year = {2025}, eprint = {2507.00505}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2507.00505v1}, }