@article{reinforcementlearningbasedknowledgedisti, title = {Reinforcement Learning-based Knowledge Distillation with LLM-as-a-Judge}, author = {Yiyang Shen and Lifu Tu and Weiran Wang}, year = {2026}, eprint = {2604.02621}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2604.02621}, }