@article{learningpoliciesfromselfplaywithpolicy, title = {Learning Policies from Self-Play with Policy Gradients and MCTS Value Estimates}, author = {Dennis J. N. J. Soemers and Éric Piette and Matthew Stephenson and Cameron Browne}, year = {2019}, eprint = {1905.05809}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/1905.05809v1}, }