@article{learningfromfailurescorrectionorientedpo, title = {Learning from Failures: Correction-Oriented Policy Optimization with Verifiable Rewards}, author = {Mengjie Ren and Jie Lou and Boxi Cao and Xueru Wen and Hongyu Lin and Xianpei Han and Le Sun and Xing Yu and Yaojie Lu}, year = {2026}, eprint = {2605.14539}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2605.14539}, }