@article{alternatingreinforcementlearningforrubri, title = {Alternating Reinforcement Learning for Rubric-Based Reward Modeling in Non-Verifiable LLM Post-Training}, author = {Ran Xu and Tianci Liu and Zihan Dong and Tony Yu and Ilgee Hong and Carl Yang and Linjun Zhang and Tao Zhao and Haoyu Wang}, year = {2026}, eprint = {2602.01511}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2602.01511}, }