@article{robustposttrainingforgenerativerecommend, title = {Robust Post-Training for Generative Recommenders: Why Exponential Reward-Weighted SFT Outperforms RLHF}, author = {Keertana Chidambaram and Sanath Kumar Krishnamurthy and Qiuling Xu and Ko-Jen Hsiao and Moumita Bhattacharya}, year = {2026}, eprint = {2603.10279}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2603.10279}, }