@article{perceptionencoderthebestvisualembeddings, title = {Perception Encoder: The best visual embeddings are not at the output of the network}, author = {Daniel Bolya and Po-Yao Huang and Peize Sun and Jang Hyun Cho and Andrea Madotto and Chen Wei and Tengyu Ma and Jiale Zhi and Jathushan Rajasegaran and Hanoona Rasheed and Junke Wang and Marco Monteiro and Hu Xu and Shiyu Dong and Nikhila Ravi and Daniel Li and Piotr Dollár and Christoph Feichtenhofer}, year = {2025}, eprint = {2504.13181}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2504.13181v1}, }