@article{rethinkingreinforcementfinetuningofllmsa, title = {Rethinking Reinforcement fine-tuning of LLMs: A Multi-armed Bandit Learning Perspective}, author = {Xiao Hu and Hong Xie and Tao Tan and Defu Lian and Jianyu Han}, year = {2026}, eprint = {2601.14599}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2601.14599}, }