@article{selfplaywithexecutionfeedbackimproving, title = {Self-play with Execution Feedback: Improving Instruction-following Capabilities of Large Language Models}, author = {Guanting Dong and Keming Lu and Chengpeng Li and Tingyu Xia and Bowen Yu and Chang Zhou and Jingren Zhou}, year = {2024}, eprint = {2406.13542}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2406.13542v3}, }