@article{trolltrustregionsimprovereinforcementlea, title = {TROLL: Trust Regions improve Reinforcement Learning for Large Language Models}, author = {Philipp Becker and Niklas Freymuth and Serge Thilges and Fabian Otto and Gerhard Neumann}, year = {2025}, eprint = {2510.03817}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2510.03817}, }