@article{learningpolicyfromasingletrajectoryinave, title = {Learning Policy from a Single Trajectory in Average-Reward Markov Decision Process}, author = {Jongmin Lee and Ernest K. Ryu and Vaneet Aggarwal}, year = {2026}, eprint = {2606.16729}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2606.16729}, }