@article{learningtoshardrlforcooptimizingtheparal, title = {Learning to Shard: RL for Co-optimizing the Parallelism Degrees and Per-operator Sharding Dimensions in Distributed LLM Inference}, author = {Ruokai Yin and Sattwik Deb Mishra and Xuan Zuo and Hokchhay Tann and Preyas Shah and Apala Guha}, year = {2025}, eprint = {2509.00217}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2509.00217}, }