@article{xiaoicetrainingfreevideounderstandingvia, title = {Xiaoice: Training-Free Video Understanding via Self-Supervised Spatio-Temporal Clustering of Semantic Features}, author = {Shihao Ji and Zihui Song}, year = {2025}, eprint = {2510.16781}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2510.16781}, }