@article{stairaddressingstagemisalignmentthrought, title = {STAIR: Addressing Stage Misalignment through Temporal-Aligned Preference Reinforcement Learning}, author = {Yao Luan and Ni Mu and Yiqin Yang and Bo Xu and Qing-Shan Jia}, year = {2025}, eprint = {2509.23802}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2509.23802}, }