@article{canllmsguidetheirownexplorationgradientg, title = {Can LLMs Guide Their Own Exploration? Gradient-Guided Reinforcement Learning for LLM Reasoning}, author = {Zhenwen Liang and Sidi Lu and Wenhao Yu and Kishan Panaganti and Yujun Zhou and Haitao Mi and Dong Yu}, year = {2025}, eprint = {2512.15687}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2512.15687}, }