@article{understandingreinforcementlearningformod, title = {Understanding Reinforcement Learning for Model Training, and future directions with GRAPE}, author = {Rohit Patel}, year = {2025}, eprint = {2509.04501}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2509.04501}, }