@article{reinforcementlearningwithoutgroundtruths, title = {Reinforcement Learning without Ground-Truth Solutions can Improve LLMs}, author = {Yingyu Lin and Qiyue Gao and Nikki Lijing Kuang and Xunpeng Huang and Kun Zhou and Tongtong Liang and Zhewei Yao and Yi-An Ma and Yuxiong He}, year = {2026}, eprint = {2606.27369}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2606.27369}, }