@article{visionlanguagevisionautoencoderscalablek, title = {Vision-Language-Vision Auto-Encoder: Scalable Knowledge Distillation from Diffusion Models}, author = {Tiezheng Zhang and Yitong Li and Yu-cheng Chou and Jieneng Chen and Alan Yuille and Chen Wei and Junfei Xiao}, year = {2025}, eprint = {2507.07104}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2507.07104}, }